<?xml version="1.0"?>
<feed xmlns="http://www.w3.org/2005/Atom" xml:lang="en">
	<id>https://wiki.inria.fr/wikis/popix/api.php?action=feedcontributions&amp;feedformat=atom&amp;user=Admin</id>
	<title>Popix - User contributions [en]</title>
	<link rel="self" type="application/atom+xml" href="https://wiki.inria.fr/wikis/popix/api.php?action=feedcontributions&amp;feedformat=atom&amp;user=Admin"/>
	<link rel="alternate" type="text/html" href="https://wiki.inria.fr/popix/Special:Contributions/Admin"/>
	<updated>2026-09-18T22:11:00Z</updated>
	<subtitle>User contributions</subtitle>
	<generator>MediaWiki 1.32.6</generator>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7481</id>
		<title>MediaWiki:Vector.css</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7481"/>
		<updated>2014-04-11T06:52:37Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;/* Le CSS placé ici affectera les utilisateurs de l’habillage Vector. */&lt;br /&gt;
@import &amp;quot;https://commons.inria.fr/wiki/vector.css&amp;quot;;&lt;br /&gt;
div#mw-head {&lt;br /&gt;
   background-image:url(https://commons.inria.fr/wiki/images/titres/mwhead-popix.jpg);&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
body { &lt;br /&gt;
font-size:12pt;&lt;br /&gt;
font-family:garamond;&lt;br /&gt;
}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7480</id>
		<title>MediaWiki:Vector.css</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7480"/>
		<updated>2014-04-03T15:03:39Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;/* Le CSS placé ici affectera les utilisateurs de l’habillage Vector. */&lt;br /&gt;
@import &amp;quot;https://commons.inria.fr/wiki/vector.css&amp;quot;;&lt;br /&gt;
div#mw-head {&lt;br /&gt;
   background-image:url(https://commons.inria.fr/wiki/images/titres/mwhead-popix.jpg);&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
#left-navigation {&lt;br /&gt;
left: 84px;&lt;br /&gt;
top: 84px;&lt;br /&gt;
}&lt;br /&gt;
div#simpleSearch button#searchButton {&lt;br /&gt;
    background-color: transparent;&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
body { &lt;br /&gt;
font-size:12pt;&lt;br /&gt;
font-family:garamond;&lt;br /&gt;
}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7479</id>
		<title>MediaWiki:Vector.css</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7479"/>
		<updated>2014-04-03T14:57:13Z</updated>

		<summary type="html">&lt;p&gt;Admin: Undo revision 7478 by Admin (talk)&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;/* Le CSS placé ici affectera les utilisateurs de l’habillage Vector. */&lt;br /&gt;
@import &amp;quot;https://commons.inria.fr/wiki/vector.css&amp;quot;;&lt;br /&gt;
div#mw-head {&lt;br /&gt;
   background-image:url(https://commons.inria.fr/wiki/images/titres/mwhead-popix.jpg);&lt;br /&gt;
}&lt;br /&gt;
body { &lt;br /&gt;
font-size:12pt;&lt;br /&gt;
font-family:garamond;&lt;br /&gt;
}&lt;br /&gt;
#left-navigation {&lt;br /&gt;
left: 84px;&lt;br /&gt;
top: 84px;&lt;br /&gt;
}&lt;br /&gt;
div#simpleSearch button#searchButton {&lt;br /&gt;
    background-color: transparent;&lt;br /&gt;
}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7478</id>
		<title>MediaWiki:Vector.css</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7478"/>
		<updated>2014-04-03T14:56:10Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;/* Le CSS placé ici affectera les utilisateurs de l’habillage Vector. */&lt;br /&gt;
@import &amp;quot;https://commons.inria.fr/wiki/vector.css&amp;quot;;&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7477</id>
		<title>MediaWiki:Vector.css</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7477"/>
		<updated>2014-04-03T14:20:11Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;/* Le CSS placé ici affectera les utilisateurs de l’habillage Vector. */&lt;br /&gt;
@import &amp;quot;https://commons.inria.fr/wiki/vector.css&amp;quot;;&lt;br /&gt;
div#mw-head {&lt;br /&gt;
   background-image:url(https://commons.inria.fr/wiki/images/titres/mwhead-popix.jpg);&lt;br /&gt;
}&lt;br /&gt;
body { &lt;br /&gt;
font-size:12pt;&lt;br /&gt;
font-family:garamond;&lt;br /&gt;
}&lt;br /&gt;
#left-navigation {&lt;br /&gt;
left: 84px;&lt;br /&gt;
top: 84px;&lt;br /&gt;
}&lt;br /&gt;
div#simpleSearch button#searchButton {&lt;br /&gt;
    background-color: transparent;&lt;br /&gt;
}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7476</id>
		<title>MediaWiki:Vector.css</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=MediaWiki:Vector.css&amp;diff=7476"/>
		<updated>2014-04-03T14:19:13Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;/* Le CSS placé ici affectera les utilisateurs de l’habillage Vector. */&lt;br /&gt;
&lt;br /&gt;
body { &lt;br /&gt;
font-size:12pt;&lt;br /&gt;
font-family:garamond;&lt;br /&gt;
}&lt;br /&gt;
#left-navigation {&lt;br /&gt;
left: 84px;&lt;br /&gt;
top: 84px;&lt;br /&gt;
}&lt;br /&gt;
div#simpleSearch button#searchButton {&lt;br /&gt;
    background-color: transparent;&lt;br /&gt;
}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7475</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7475"/>
		<updated>2013-08-28T13:58:54Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Selecting the error model */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
phi1=psi1[c(1,2,3)]&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
phi2=psi2[c(1,2,3)]&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
          ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;first order absorption&amp;quot;, &lt;br /&gt;
          &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
phi3=psi3[c(1,2,3)]&lt;br /&gt;
fc3=predc2(tc,phi3)&lt;br /&gt;
phi4=psi4[c(1,2,3)]&lt;br /&gt;
fc4=predc2(tc,phi4)&lt;br /&gt;
phi5=psi5[c(1,2,3)]&lt;br /&gt;
fc5=predc2(tc,phi5)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;,ylab=&amp;quot;concentration (mg/l)&amp;quot;,&lt;br /&gt;
        col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc3, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;cyan&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc5, type = &amp;quot;l&amp;quot;, col = &amp;quot;magenta&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;constant error model&amp;quot;,&lt;br /&gt;
        &amp;quot;proportional error model&amp;quot;,&amp;quot;combined error model&amp;quot;,&amp;quot;exponential error model&amp;quot;),&lt;br /&gt;
 lty=c(-1,1,1,1,1), pch=c(1,-1,-1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;cyan&amp;quot;,&amp;quot;magenta&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
ci4=matrix(c(psi4-delta4,psi4+delta4),ncol=2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H){&lt;br /&gt;
  dphi=length(phi)&lt;br /&gt;
  nf=length(f)&lt;br /&gt;
  H=H*n/(n-dphi)&lt;br /&gt;
  S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
  G=matrix(nrow=nf,ncol=dphi)&lt;br /&gt;
  for (k in seq(1,dphi)) {&lt;br /&gt;
    dk=phi[k]*(1e-5)&lt;br /&gt;
    phid=phi&lt;br /&gt;
    phid[k]=phi[k] + dk&lt;br /&gt;
    fd=predc2(tc,phid)&lt;br /&gt;
    G[,k]=(f-fd)/dk&lt;br /&gt;
  }&lt;br /&gt;
  M=rowSums((G%*%S)*G)&lt;br /&gt;
  deltaf=sqrt(M)*qt(0.5+alpha/2,df)&lt;br /&gt;
return(deltaf)}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
       lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7474</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7474"/>
		<updated>2013-08-28T13:52:41Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Selecting the error model */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
phi1=psi1[c(1,2,3)]&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
phi2=psi2[c(1,2,3)]&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
          ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;first order absorption&amp;quot;, &lt;br /&gt;
          &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
phi3=psi3[c(1,2,3)]&lt;br /&gt;
fc3=predc2(tc,phi3)&lt;br /&gt;
phi4=psi4[c(1,2,3)]&lt;br /&gt;
fc4=predc2(tc,phi4)&lt;br /&gt;
phi5=psi5[c(1,2,3)]&lt;br /&gt;
fc5=predc2(tc,phi5)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;,ylab=&amp;quot;concentration (mg/l)&amp;quot;,&lt;br /&gt;
        col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc3, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;cyan&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc5, type = &amp;quot;l&amp;quot;, col = &amp;quot;magenta&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;constant error model&amp;quot;,&lt;br /&gt;
        &amp;quot;proportional error model&amp;quot;,&amp;quot;combined error model&amp;quot;,&amp;quot;exponential error model&amp;quot;),&lt;br /&gt;
 lty=c(-1,1,1,1,1), pch=c(1,-1,-1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;cyan&amp;quot;,&amp;quot;magenta&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
ci4=matrix(c(psi4-delta4,psi4+delta4),ncol=2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H){&lt;br /&gt;
  dphi=length(phi)&lt;br /&gt;
  nf=length(f)&lt;br /&gt;
  H=H*n/(n-dphi)&lt;br /&gt;
  S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
  G=matrix(nrow=nf,ncol=dphi)&lt;br /&gt;
  for (k in seq(1,dphi)) {&lt;br /&gt;
    dk=phi[k]*(1e-5)&lt;br /&gt;
    phid=phi&lt;br /&gt;
    phid[k]=phi[k] + dk&lt;br /&gt;
    fd=predc2(tc,phid)&lt;br /&gt;
    G[,k]=(f-fd)/dk&lt;br /&gt;
  }&lt;br /&gt;
  M=rowSums((G%*%S)*G)&lt;br /&gt;
  deltaf=sqrt(M)*qt(0.5+alpha/2,df)&lt;br /&gt;
return(deltaf)}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7473</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7473"/>
		<updated>2013-08-28T13:47:17Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Selecting the error model */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
phi1=psi1[c(1,2,3)]&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
phi2=psi2[c(1,2,3)]&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
          ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;first order absorption&amp;quot;, &lt;br /&gt;
          &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
phi3=psi3[c(1,2,3)]&lt;br /&gt;
fc3=predc2(tc,phi3)&lt;br /&gt;
phi4=psi4[c(1,2,3)]&lt;br /&gt;
fc4=predc2(tc,phi4)&lt;br /&gt;
phi5=psi5[c(1,2,3)]&lt;br /&gt;
fc5=predc2(tc,phi5)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;,ylab=&amp;quot;concentration (mg/l)&amp;quot;,&lt;br /&gt;
        col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc3, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;cyan&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc5, type = &amp;quot;l&amp;quot;, col = &amp;quot;magenta&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;constant error model&amp;quot;,&lt;br /&gt;
        &amp;quot;proportional error model&amp;quot;,&amp;quot;combined error model&amp;quot;,&amp;quot;exponential error model&amp;quot;),&lt;br /&gt;
 lty=c(-1,1,1,1,1), pch=c(1,-1,-1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;cyan&amp;quot;,&amp;quot;magenta&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
ci4=matrix(c(psi4-delta4,psi4+delta4),ncol=2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H)&lt;br /&gt;
{&lt;br /&gt;
dphi=length(phi)&lt;br /&gt;
nf=length(f)&lt;br /&gt;
H=H*n/(n-dphi)&lt;br /&gt;
S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
G=matrix(nrow=nf, ncol=dphi)&lt;br /&gt;
for (k in seq(1,dphi)) {&lt;br /&gt;
   dk=phi[k]*(1e-5)&lt;br /&gt;
   phid=phi&lt;br /&gt;
   phid[k]=phi[k] + dk&lt;br /&gt;
   fd=predc2(tc,phid)&lt;br /&gt;
   G[,k]=(f-fd)/dk&lt;br /&gt;
}&lt;br /&gt;
M=rowSums((G%*%S)*G)&lt;br /&gt;
deltaf=sqrt(M)*qt(0.5+ialpha/2,df)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7472</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7472"/>
		<updated>2013-08-28T13:45:17Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Selecting the error model */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
phi1=psi1[c(1,2,3)]&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
phi2=psi2[c(1,2,3)]&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
          ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;first order absorption&amp;quot;, &lt;br /&gt;
          &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
phi3=psi3[c(1,2,3)]&lt;br /&gt;
fc3=predc2(tc,phi3)&lt;br /&gt;
phi4=psi4[c(1,2,3)]&lt;br /&gt;
fc4=predc2(tc,phi4)&lt;br /&gt;
phi5=psi5[c(1,2,3)]&lt;br /&gt;
fc5=predc2(tc,phi5)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;,ylab=&amp;quot;concentration (mg/l)&amp;quot;,&lt;br /&gt;
        col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc3, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;cyan&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc5, type = &amp;quot;l&amp;quot;, col = &amp;quot;magenta&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;constant error model&amp;quot;,&lt;br /&gt;
        &amp;quot;proportional error model&amp;quot;,&amp;quot;combined error model&amp;quot;,&amp;quot;exponential error model&amp;quot;),&lt;br /&gt;
 lty=c(-1,1,1,1,1), pch=c(1,-1,-1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;cyan&amp;quot;,&amp;quot;magenta&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H)&lt;br /&gt;
{&lt;br /&gt;
dphi=length(phi)&lt;br /&gt;
nf=length(f)&lt;br /&gt;
H=H*n/(n-dphi)&lt;br /&gt;
S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
G=matrix(nrow=nf, ncol=dphi)&lt;br /&gt;
for (k in seq(1,dphi)) {&lt;br /&gt;
   dk=phi[k]*(1e-5)&lt;br /&gt;
   phid=phi&lt;br /&gt;
   phid[k]=phi[k] + dk&lt;br /&gt;
   fd=predc2(tc,phid)&lt;br /&gt;
   G[,k]=(f-fd)/dk&lt;br /&gt;
}&lt;br /&gt;
M=rowSums((G%*%S)*G)&lt;br /&gt;
deltaf=sqrt(M)*qt(0.5+ialpha/2,df)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7471</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7471"/>
		<updated>2013-08-28T13:35:57Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Fitting different error models */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
phi1=psi1[c(1,2,3)]&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
phi2=psi2[c(1,2,3)]&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
          ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;first order absorption&amp;quot;, &lt;br /&gt;
          &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
        ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;, &lt;br /&gt;
        &amp;quot;first order absorption&amp;quot;,&lt;br /&gt;
        &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H)&lt;br /&gt;
{&lt;br /&gt;
dphi=length(phi)&lt;br /&gt;
nf=length(f)&lt;br /&gt;
H=H*n/(n-dphi)&lt;br /&gt;
S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
G=matrix(nrow=nf, ncol=dphi)&lt;br /&gt;
for (k in seq(1,dphi)) {&lt;br /&gt;
   dk=phi[k]*(1e-5)&lt;br /&gt;
   phid=phi&lt;br /&gt;
   phid[k]=phi[k] + dk&lt;br /&gt;
   fd=predc2(tc,phid)&lt;br /&gt;
   G[,k]=(f-fd)/dk&lt;br /&gt;
}&lt;br /&gt;
M=rowSums((G%*%S)*G)&lt;br /&gt;
deltaf=sqrt(M)*qt(0.5+ialpha/2,df)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7470</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7470"/>
		<updated>2013-08-28T13:34:08Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Assessing and selecting the PK model */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
phi1=psi1[c(1,2,3)]&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
phi2=psi2[c(1,2,3)]&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
          ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;first order absorption&amp;quot;, &lt;br /&gt;
          &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]&lt;br /&gt;
e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
        ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;, &lt;br /&gt;
        &amp;quot;first order absorption&amp;quot;,&lt;br /&gt;
        &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H)&lt;br /&gt;
{&lt;br /&gt;
dphi=length(phi)&lt;br /&gt;
nf=length(f)&lt;br /&gt;
H=H*n/(n-dphi)&lt;br /&gt;
S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
G=matrix(nrow=nf, ncol=dphi)&lt;br /&gt;
for (k in seq(1,dphi)) {&lt;br /&gt;
   dk=phi[k]*(1e-5)&lt;br /&gt;
   phid=phi&lt;br /&gt;
   phid[k]=phi[k] + dk&lt;br /&gt;
   fd=predc2(tc,phid)&lt;br /&gt;
   G[,k]=(f-fd)/dk&lt;br /&gt;
}&lt;br /&gt;
M=rowSums((G%*%S)*G)&lt;br /&gt;
deltaf=sqrt(M)*qt(0.5+ialpha/2,df)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7469</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7469"/>
		<updated>2013-08-28T13:33:34Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Assessing and selecting the PK model */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
phi1=psi1[c(1,2,3)]&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
phi2=psi2[c(1,2,3)]&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,&amp;quot;first order absorption&amp;quot;, &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]&lt;br /&gt;
e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
        ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;, &lt;br /&gt;
        &amp;quot;first order absorption&amp;quot;,&lt;br /&gt;
        &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H)&lt;br /&gt;
{&lt;br /&gt;
dphi=length(phi)&lt;br /&gt;
nf=length(f)&lt;br /&gt;
H=H*n/(n-dphi)&lt;br /&gt;
S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
G=matrix(nrow=nf, ncol=dphi)&lt;br /&gt;
for (k in seq(1,dphi)) {&lt;br /&gt;
   dk=phi[k]*(1e-5)&lt;br /&gt;
   phid=phi&lt;br /&gt;
   phid[k]=phi[k] + dk&lt;br /&gt;
   fd=predc2(tc,phid)&lt;br /&gt;
   G[,k]=(f-fd)/dk&lt;br /&gt;
}&lt;br /&gt;
M=rowSums((G%*%S)*G)&lt;br /&gt;
deltaf=sqrt(M)*qt(0.5+ialpha/2,df)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7468</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7468"/>
		<updated>2013-08-28T13:32:09Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Fitting two PK models */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t){&lt;br /&gt;
  f=predc1(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t){&lt;br /&gt;
  f=predc2(t,x)&lt;br /&gt;
  g=x[4]&lt;br /&gt;
  e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
return(e)}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,  &amp;quot;first order absorption&amp;quot;,&lt;br /&gt;
       &amp;quot;zero order absorption&amp;quot;), lty=c(-1,1,1), &lt;br /&gt;
       pch=c(1,-1,-1), lwd=2,&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]&lt;br /&gt;
e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
        ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;, &lt;br /&gt;
        &amp;quot;first order absorption&amp;quot;,&lt;br /&gt;
        &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H)&lt;br /&gt;
{&lt;br /&gt;
dphi=length(phi)&lt;br /&gt;
nf=length(f)&lt;br /&gt;
H=H*n/(n-dphi)&lt;br /&gt;
S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
G=matrix(nrow=nf, ncol=dphi)&lt;br /&gt;
for (k in seq(1,dphi)) {&lt;br /&gt;
   dk=phi[k]*(1e-5)&lt;br /&gt;
   phid=phi&lt;br /&gt;
   phid[k]=phi[k] + dk&lt;br /&gt;
   fd=predc2(tc,phid)&lt;br /&gt;
   G[,k]=(f-fd)/dk&lt;br /&gt;
}&lt;br /&gt;
M=rowSums((G%*%S)*G)&lt;br /&gt;
deltaf=sqrt(M)*qt(0.5+ialpha/2,df)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7467</id>
		<title>The individual approach</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=The_individual_approach&amp;diff=7467"/>
		<updated>2013-08-28T13:30:59Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Fitting two PK models */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;br /&gt;
== Overview ==&lt;br /&gt;
&lt;br /&gt;
Before we start looking at modeling a whole population at the same time, we are going to consider only one individual from that population. Much of the basic methodology for modeling one individual follows through to population modeling. We will see that when stepping up from one individual to a population, the difference is that some parameters shared by individuals are considered to be drawn from a [http://en.wikipedia.org/wiki/Probability_distribution probability distribution].&lt;br /&gt;
&lt;br /&gt;
Let us begin with a simple  example.&lt;br /&gt;
An individual receives 100mg of a drug at time $t=0$. At that time and then every hour for fifteen hours, the&lt;br /&gt;
concentration of a marker in the bloodstream is measured and plotted against time:&lt;br /&gt;
&lt;br /&gt;
::[[File:New_Individual1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
We aim to find a mathematical model to describe what we see in the figure. The eventual goal is then to extend this approach to the ''simultaneous modeling'' of a whole population.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model and methods for the individual approach ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Defining a model===&lt;br /&gt;
&lt;br /&gt;
In our example, the concentration is a ''continuous'' variable, so we will  try to use continuous functions to model it.&lt;br /&gt;
Different types of data  (e.g., [http://en.wikipedia.org/wiki/Count_data count data], [http://en.wikipedia.org/wiki/Categorical_data categorical data], [http://en.wikipedia.org/wiki/Survival_analysis time-to-event data], etc.) require different types of models. All of these data types will be considered in due time, but for now let us concentrate on a continuous data model.&lt;br /&gt;
&lt;br /&gt;
A model for continuous data can be represented mathematically as follows:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + e_j, \quad \quad  1\leq j \leq n, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $f$ is called the ''structural model''. It corresponds to the basic type of curve we suspect the data is following, e.g., linear, logarithmic, exponential, etc. Sometimes, a model of the associated biological processes leads to equations that define the curve's shape.&lt;br /&gt;
&lt;br /&gt;
* $(t_1,t_2,\ldots , t_n)$  is the vector of observation times. Here, $t_1 = 0$ hours and $t_n = t_{16} = 15$ hours.&lt;br /&gt;
&lt;br /&gt;
* $\psi=(\psi_1, \psi_2, \ldots, \psi_d)$   is a vector of $d$ parameters that influences the value of $f$.&lt;br /&gt;
&lt;br /&gt;
* $(e_1, e_2, \ldots, e_n)$  are called the ''residual errors''. Usually, we suppose that they come from some centered probability distribution: $\esp{e_j} =0$. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In fact, we usually state a continuous data model in a slightly more flexible way:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;cont&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{j} = f(t_j ; \psi) + g(t_j ; \psi)\teps_j  , \quad \quad  1\leq j \leq n,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
where now:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $g$  is called the ''residual error model''. It may be a function of the time $t_j$ and parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
* $(\teps_1, \teps_2, \ldots, \teps_n)$  are the ''normalized'' residual errors. We suppose that these come from a probability distribution which is centered and has unit variance: $\esp{\teps_j} = 0$ and $\var{\teps_j} =1$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Choosing a residual error model===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The choice of a residual error model $g$ is very flexible, and allows us to account for many different hypotheses we may have on the error's distribution. Let $f_j=f(t_j;\psi)$. Here are some simple error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant error model'': $g=a$. That is,  $y_j=f_j+a\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional error model'': $g=b\,f$.  That is, $y_j=f_j+bf_j\teps_j$. This is for when we think the magnitude of the error is proportional to the value of the predicted value $f$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Combined error model'': $g=a+b f$. Here, $y_j=f_j+(a+bf_j)\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Alternative combined error model'': $g^2=a^2+b^2f^2$. Here, $y_j=f_j+\sqrt{a^2+b^2f_j^2}\teps_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Exponential error model'': here, the model is instead $\log(y_j)=\log(f_j) + a\teps_j$, that is, $g=a$. It is exponential in the sense that if we exponentiate, we end up with $y_j = f_j e^{a\teps_j}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Tasks===&lt;br /&gt;
&lt;br /&gt;
To model a vector of observations $y = (y_j,\, 1\leq j \leq n$) we must perform several tasks:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Select a structural model $f$ and a residual error model $g$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Estimate the model's parameters $\psi$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Assess and validate'' the selected model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Selecting structural and residual error models ===&lt;br /&gt;
&lt;br /&gt;
As we are interested in [http://en.wikipedia.org/wiki/Parametric_model parametric modeling], we must choose parametric structural and residual error models. In the absence of biological (or other) information, we might suggest possible structural models just by looking at the graphs of time-evolution of the data. For example, if $y_j$ is increasing with time, we might suggest an affine, quadratic or logarithmic model, depending on the approximate trend of the data. If $y_j$ is instead decreasing ever slower to zero, an exponential model might be appropriate.&lt;br /&gt;
&lt;br /&gt;
However, often  we have biological (or other) information to help us make our choice. For instance, if we have a system of [http://en.wikipedia.org/wiki/Differential_equation differential equations] describing how the drug is eliminated from the body, its solution may provide the formula (i.e., structural model) we are looking for.&lt;br /&gt;
&lt;br /&gt;
As for the residual error model, if it is not immediately obvious which one to choose, several can be tested in conjunction with one or several possible structural models. After parameter estimation, each structural and residual error model pair can be assessed, compared against the others, and/or validated in various ways.&lt;br /&gt;
&lt;br /&gt;
Now we can have a first look at parameter estimation, and further on, model assessment and validation.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Parameter estimation===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Given the observed data and the choice of a parametric model to describe it, our goal becomes to find the &amp;quot;best&amp;quot; parameters for the model. A traditional framework to solve this kind of problem is called [http://en.wikipedia.org/wiki/Maximum_likelihood maximum likelihood estimation] or MLE, in which the &amp;quot;most likely&amp;quot; parameters are found, given the data that was observed.&lt;br /&gt;
&lt;br /&gt;
The likelihood $L$ is a function defined as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; L(\psi ; y_1,y_2,\ldots,y_n) \ \ \eqdef \ \ \py( y_1,y_2,\ldots,y_n; \psi) , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
i.e., the conditional [http://en.wikipedia.org/wiki/Joint_probability_distribution joint density function] of $(y_j)$ given the parameters $\psi$, but looked at as if the data are known and the parameters not. The $\hat{\psi}$ which maximizes $L$ is known as the ''maximum likelihood estimator''.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have chosen a structural model $f$ and residual error model $g$. If we assume for instance that $ \teps_j \sim_{i.i.d} {\cal N}(0,1)$, then the $y_j$ are independent of each other and [[#cont|(1)]] means that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y_{j} \sim {\cal N}\left(f(t_j ; \psi) , g(t_j ; \psi)^2\right), \quad \quad  1\leq j \leq n .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Due to this independence, the pdf of $y = (y_1, y_2, \ldots, y_n)$ is the product of the pdfs of each $y_j$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\py(y_1, y_2, \ldots y_n ; \psi) &amp;amp;=&amp;amp; \prod_{j=1}^n \pyj(y_j ; \psi) \\ \\&lt;br /&gt;
&amp;amp; = &amp;amp;  \frac{1}{\prod_{j=1}^n \sqrt{2\pi} g(t_j ; \psi)} \   {\rm exp}\left\{-\frac{1}{2} \sum_{j=1}^n \left( \displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2\right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This is the same thing as the likelihood function $L$ when seen as a function of $\psi$. Maximizing $L$ is equivalent to minimizing the deviance, i.e., -2 $\times$ the $\log$-likelihood ($LL$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;LLL&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\psi} &amp;amp;=&amp;amp;   \argmin{\psi} \left\{ -2 \,LL \right\}\\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmin{\psi} \left\{&lt;br /&gt;
\sum_{j=1}^n \log\left(g(t_j ; \psi)^2\right)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \psi)}{g(t_j ; \psi)} }\right)^2 \right\} . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This minimization problem does not usually have an [http://en.wikipedia.org/wiki/Analytical_expression analytical solution] for nonlinear models, so an [http://en.wikipedia.org/wiki/Mathematical_optimization optimization] procedure needs to be used.&lt;br /&gt;
However, for a few specific models, analytical solutions do exist.&lt;br /&gt;
&lt;br /&gt;
For instance, suppose we have a constant error model: $y_{j} = f(t_j ; \psi)  + a \, \teps_j,\,\,  1\leq j \leq n,$ that is: $g(t_j;\psi) = a$. In practice, $f$ is not itself a function of $a$, so we can write $\psi = (\phi,a)$ and therefore: $y_{j} = f(t_j ; \phi)  + a \, \teps_j.$ Thus, [[#LLL|(2)]] simplifies to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; (\hat{\phi},\hat{a}) \ \ = \ \ \argmin{(\phi,a)} \left\{&lt;br /&gt;
n \log(a^2)  + \sum_{j=1}^n \left(\displaystyle{ \frac{y_j - f(t_j ; \phi)}{a} }\right)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The solution is then:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; \argmin{\phi}  \sum_{j=1}^n \left( y_j - f(t_j ; \phi)\right)^2 \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp;  \frac{1}{n}\sum_{j=1}^n \left( y_j - f(t_j ; \hat{\phi})\right)^2 ,&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{a}^2$ is found by setting the [http://en.wikipedia.org/wiki/Partial_derivative partial derivative] of $-2LL$ to zero.&lt;br /&gt;
&lt;br /&gt;
Whether this has an analytical solution or not depends on the form of $f$. For example, if $f(t_j;\phi)$ is just a linear function of the components of the vector $\phi$, we can represent it as a matrix $F$ whose $j$th row gives the coefficients at time $t_j$. Therefore, we have the matrix equation $y = F \phi + a \teps$.&lt;br /&gt;
&lt;br /&gt;
The solution for $\hat{\phi}$ is thus the least-squares one, and for $\hat{a}^2$ it is the same as before:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\phi} &amp;amp;=&amp;amp; (F^\prime F)^{-1} F^\prime y \\&lt;br /&gt;
\hat{a}^2&amp;amp;=&amp;amp; \frac{1}{n}\sum_{j=1}^n \left( y_j - F_j \hat{\phi}\right)^2 . \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Computing the Fisher information matrix===&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Fisher_information Fisher information] is a way of measuring the amount of information that an observable random variable carries about an unknown parameter upon which its probability distribution depends.&lt;br /&gt;
&lt;br /&gt;
Let $\psis $ be the true unknown value of $\psi$, and let $\hatpsi$ be the maximum likelihood estimate of $\psi$. If the observed likelihood function is sufficiently smooth, asymptotic theory for maximum-likelihood estimation holds and&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;intro_individualCLT&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
I_n(\psis)^{\frac{1}{2} }(\hatpsi-\psis) \limite{n\to \infty}{} {\mathcal N}(0,\id) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $I_n(\psis)$ is (minus) the Hessian (i.e., the matrix of the second derivatives) of the log-likelihood:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;I_n(\psis)=-  \displaystyle{ \frac{\partial^2}{\partial \psi \partial \psi^\prime} } LL(\psis;y_1,y_2,\ldots,y_n)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
is the ''observed Fisher information matrix''. Here, &amp;quot;observed&amp;quot; means that it is a function of observed variables $y_1,y_2,\ldots,y_n$.&lt;br /&gt;
&lt;br /&gt;
Thus, an estimate of the covariance of $\hatpsi$ is the inverse of the observed Fisher information matrix as expressed by the formula:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;C(\hatpsi) = - I_n(\hatpsi)^{-1} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for parameters===&lt;br /&gt;
&lt;br /&gt;
Let $\psi_k$ be the $k$th of $d$ components of $\psi$. Imagine that we have estimated $\psi_k$ with $\hatpsi_k$, the $k$th component of the MLE $\hatpsi$, that is, a random variable that converges to $\psi_k^{\star}$ when $n \to \infty$ under very general conditions.&lt;br /&gt;
&lt;br /&gt;
An estimator of its variance is the $k$th element of the diagonal of the covariance matrix $C(\hatpsi)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm Var}(\hatpsi_k) = C_{kk}(\hatpsi) .&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can thus derive an estimator of its [http://en.wikipedia.org/wiki/Standard_error standard error]:&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(\hatpsi_k) = \sqrt{C_{kk}(\hatpsi)} ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a [http://en.wikipedia.org/wiki/Confidence_interval confidence interval] of level $1-\alpha$ for $\psi_k^\star$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(\psi_k^\star) = \left[\hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(\frac{\alpha}{2}\right), \ \hatpsi_k + \widehat{\rm s.e.}(\hatpsi_k)\,q\left(1-\frac{\alpha}{2}\right)\right] , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $q(w)$ is the [http://en.wikipedia.org/wiki/Quantile quantile] of order $w$ of a ${\cal N}(0,1)$ distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Approximating the fraction $\hatpsi/\widehat{\rm s.e}(\hatpsi_k)$ by the normal distribution is a &amp;quot;good&amp;quot; approximation only when the number of observations $n$ is large. A better approximation should be used for small $n$. In the model $y_j = f(t_j ; \phi) + a\teps_j$, the distribution of $\hat{a}^2$ can be approximated by a [http://en.wikipedia.org/wiki/Chi-squared_distribution chi-squared  distribution] with $(n-d_\phi)$ [http://en.wikipedia.org/wiki/Degrees_of_freedom_%28statistics%29 degrees of freedom], where $d_\phi$ is the dimension of $\phi$. The quantiles of the normal distribution can then be replaced by those of a [http://en.wikipedia.org/wiki/Student%27s_t-distribution Student's $t$-distribution] with $(n-d_\phi)$ degrees of freedom.&lt;br /&gt;
&amp;lt;!-- %$${\rm CI}(\psi_k) = [\hatpsi_k - \widehat{\rm s.e}(\hatpsi_k)q((1-\alpha)/2,n-d) , \hatpsi_k + \widehat{\rm s.e}(\hatpsi_k)q((1+\alpha)/2,n-d)]$$ --&amp;gt;&lt;br /&gt;
&amp;lt;!--  %where $q(\alpha,\nu)$ is the quantile of order $\alpha$ of a $t$-distribution with $\nu$ degrees of freedom. --&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Deriving confidence intervals for predictions===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model $f$ can be predicted for any $t$ using the estimated value $f(t; \hatphi)$. For that $t$, we can then derive a confidence interval for $f(t,\phi)$ using the estimated variance of $\hatphi$. Indeed, as a first approximation we have:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; f(t ; \hatphi) \simeq f(t ; \phis) + \nabla f (t,\phis) (\hatphi - \phis) ,&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\nabla f(t,\phis)$ is the gradient of $f$ at $\phis$, i.e., the vector of the first-order partial derivatives of $f$ with respect to the components of $\phi$, evaluated at $\phis$. Of course, we do not actually know $\phis$, but we can estimate $\nabla f(t,\phis)$  with $\nabla f(t,\hatphi)$. The variance of $f(t ; \hatphi)$ can then be estimated by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\widehat{\rm Var}\left(f(t ; \hatphi)\right) \simeq \nabla f (t,\hatphi)\widehat{\rm Var}(\hatphi) \left(\nabla f (t,\hatphi) \right)^\prime . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then derive an estimate of the standard error of $f (t,\hatphi)$ for any $t$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\widehat{\rm s.e.}(f(t ; \hatphi)) = \sqrt{\widehat{\rm Var}\left(f(t ; \hatphi)\right)} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and a confidence interval of level $1-\alpha$ for $f(t ; \phi^\star)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;{\rm CI}(f(t ; \phi^\star)) = \left[f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(\frac{\alpha}{2}\right), \ f(t ; \hatphi) + \widehat{\rm s.e.}(f(t ; \hatphi))\,q\left(1-\frac{\alpha}{2}\right)\right].&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===Estimating confidence intervals using Monte Carlo simulation===&lt;br /&gt;
&lt;br /&gt;
The use of [http://en.wikipedia.org/wiki/Monte_Carlo_method Monte Carlo methods] to estimate a distribution does not require any approximation of the  model.&lt;br /&gt;
&lt;br /&gt;
We proceed in the following way. Suppose we have found a MLE $\hatpsi$ of $\psi$. We then simulate a data vector $y^{(1)}$ by first randomly generating the vector $\teps^{(1)}$ and then calculating for $1 \leq j \leq n$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; y^{(1)}_j = f(t_j ;\hatpsi) + g(t_j ;\hatpsi)\teps^{(1)}_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In a sense, this gives us an example of &amp;quot;new&amp;quot; data from the &amp;quot;same&amp;quot; model. We can then compute a new MLE $\hat{\psi}^{(1)}$ of $\psi$ using $y^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
Repeating this process $M$ times gives $M$ estimates of $\psi$ from which we can obtain an empirical estimation of the distribution of $\hatpsi$, or any quantile we like.&lt;br /&gt;
&lt;br /&gt;
Any confidence interval for $\psi_k$ (resp. $f(t,\psi_k)$) can then be approximated by a prediction interval for $\hatpsi_k$ (resp. $f(t,\hatpsi_k)$). For instance, a two-sided confidence interval of level  $1-\alpha$ for $\psi_k^\star$ can be estimated by the prediction interval&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; [\hat{\psi}_{k,([\frac{\alpha}{2} M])} \ , \ \hat{\psi}_{k,([ (1-\frac{\alpha}{2})M])} ], &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $[\cdot]$ denotes the [http://en.wikipedia.org/wiki/Floor_and_ceiling_functions integer part] and  $(\psi_{k,(m)},\ 1 \leq m \leq M)$ the order statistic, i.e., the parameters $(\hatpsi_k^{(m)}, 1 \leq m \leq M)$ reordered so that $\hatpsi_{k,(1)} \leq \hatpsi_{k,(2)} \leq \ldots \leq \hatpsi_{k,(M)}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==A PK  example ==&lt;br /&gt;
&lt;br /&gt;
In the real world, it is often not enough to look at the data, choose one possible model and estimate the parameters. The chosen structural model may or may not be &amp;quot;good&amp;quot; at representing the data. It may be good but the chosen residual error model bad, meaning that the overall model is poor, and so on. That is why in practice we may want to try out several structural and residual error models. After performing parameter estimation for each model, various assessment tasks can then be performed in order to conclude which model is best.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===The data===&lt;br /&gt;
&lt;br /&gt;
This modeling process is illustrated in detail in the following [http://en.wikipedia.org/wiki/Pharmacokinetics PK] example. Let us consider a dose D=50mg of a drug administered orally to a patient at time $t=0$. The concentration of the drug in the bloodstream is then measured at times $(t_j) = (0.5, 1,\,1.5,\,2,\,3,\,4,\,8,\,10,\,12,\,16,\,20,\,24).$ Here is the file {{Verbatim|individualFitting_data.txt}} with the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 30%;margin-left:15em&amp;quot;&lt;br /&gt;
!|      Time	  ||    Concentration &lt;br /&gt;
|-&lt;br /&gt;
|0.5	    ||       0.94&lt;br /&gt;
|-&lt;br /&gt;
|   1.0	    ||      1.30&lt;br /&gt;
|-&lt;br /&gt;
|   1.5	    ||       1.64&lt;br /&gt;
|-&lt;br /&gt;
|   2.0	    ||        3.38&lt;br /&gt;
|-&lt;br /&gt;
|   3.0	    ||       3.72&lt;br /&gt;
|-&lt;br /&gt;
|   4.0	    ||        3.29&lt;br /&gt;
|-&lt;br /&gt;
|   8.0	    ||       1.31&lt;br /&gt;
|-&lt;br /&gt;
|  10.0	    ||       0.80&lt;br /&gt;
|-&lt;br /&gt;
|  12.0	    ||       0.39&lt;br /&gt;
|-&lt;br /&gt;
|  16.0	    ||       0.31&lt;br /&gt;
|-&lt;br /&gt;
|  20.0	    ||       0.10&lt;br /&gt;
|-&lt;br /&gt;
|  24.0	    ||       0.09&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We are going to perform the analyses for this example with the free statistical software [http://www.r-project.org/  {{Verbatim|R}}]. First, we import the data and plot it to have a look:&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | &lt;br /&gt;
[[File:NewIndividual1.png|link=]]&lt;br /&gt;
| style=&amp;quot;width: 50%&amp;quot; | {{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
pk1=read.table(&amp;quot;individualFitting_data.txt&amp;quot;,header=T) &lt;br /&gt;
t=pk1$time  &lt;br /&gt;
y=pk1$concentration&lt;br /&gt;
plot(t, y, xlab=&amp;quot;time(hour)&amp;quot;,&lt;br /&gt;
     ylab=&amp;quot;concentration(mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)   &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting two PK models===&lt;br /&gt;
&lt;br /&gt;
We are going to consider two possible structural models that may describe the observed time-course of the concentration:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A [http://en.wikipedia.org/wiki/Multi-compartment_model#Single-compartment_model one compartment model] with first-order [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption] and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_1 &amp;amp;=&amp;amp; (k_a, V, k_e) \\&lt;br /&gt;
f_1(t ; \phi_1) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* A one compartment model with zero-order absorption and linear elimination:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\phi_2 &amp;amp;=&amp;amp; (T_{k0}, V, k_e) \\&lt;br /&gt;
f_2(t ; \phi_2) &amp;amp;=&amp;amp; \left\{  \begin{array}{ll}&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} }\left( 1- e^{-k_e \, t} \right) &amp;amp; {\rm if }\ t\leq T_{k0} \\&lt;br /&gt;
\displaystyle{ \frac{D}{V \,T_{k0} \, k_e} } \left( 1- e^{-k_e \, T_{k0} } \right)e^{-k_e \, (t- T_{k0})} &amp;amp; {\rm otherwise} .&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We define each of these functions in {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
predc1=function(t,x){&lt;br /&gt;
  f=50*x[1]/x[2]/(x[1]-x[3])*(exp(-x[3]*t)-exp(-x[1]*t))&lt;br /&gt;
return(f)}&lt;br /&gt;
&lt;br /&gt;
predc2=function(t,x){&lt;br /&gt;
  f=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*t))&lt;br /&gt;
  f[t&amp;gt;x[1]]=50/x[1]/x[2]/x[3]*(1-exp(-x[3]*x[1]))*exp(-x[3]*(t[t&amp;gt;x[1]]-x[1]))&lt;br /&gt;
return(f)} &amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
We then define two models ${\cal M}_1$ and ${\cal M}_2$ that assume (for now)  constant residual error models:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\cal M}_1  : \quad y_j &amp;amp; = &amp;amp; f_1(t_j ; \phi_1) + a_1\teps_j \\&lt;br /&gt;
{\cal M}_2  : \quad y_j &amp;amp; = &amp;amp; f_2(t_j ; \phi_2) + a_2\teps_j .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can fit these two models to our data by computing the MLE $\hatpsi_1=(\hatphi_1,\hat{a}_1)$ and $\hatpsi_2=(\hatphi_2,\hat{a}_2)$ of $\psi$  under each model:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin1=function(x,y,t)&lt;br /&gt;
{f=predc1(t,x)&lt;br /&gt;
g=x[4]&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin2=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
#--------- MLE --------------------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm1=nlm(fmin1, c(0.3,6,0.2,1), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi1=pk.nlm1$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm2=nlm(fmin2, c(3,10,0.2,4), y, t, hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi2=pk.nlm2$estimate&lt;br /&gt;
&amp;lt;/pre&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
:Here are the parameter estimation results:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi1 =&amp;quot;,psi1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi1 = 0.3240916 6.001204 0.3239337 0.4366948&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi2 =&amp;quot;,psi2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi2 = 3.203111 8.999746 0.229977 0.2555242&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Assessing and selecting the PK model===&lt;br /&gt;
&lt;br /&gt;
The estimated parameters $\hatphi_1$ and $\hatphi_2$ can then be used for computing the predicted concentrations $\hat{f}_1(t)$ and $\hat{f}_2(t)$ under both models at any time $t$. These curves can then be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:New_Individual2.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1),xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;,  &amp;quot;first order absorption&amp;quot;,&lt;br /&gt;
       &amp;quot;zero order absorption&amp;quot;), lty=c(-1,1,1), &lt;br /&gt;
       pch=c(1,-1,-1), lwd=2,&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
We clearly see that a much better fit is obtained with model ${\cal M}_2$, i.e., the one assuming a zero-order absorption process.&lt;br /&gt;
&lt;br /&gt;
Another useful goodness-of-fit plot is obtained by displaying the observations $(y_j)$ versus the predictions $\hat{y}_j=f(t_j ; \hatpsi)$ given by the models:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
[[File:individual3.png|link=]]&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f1=predc1(t,phi1)&lt;br /&gt;
f2=predc2(t,phi2)&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,2))&lt;br /&gt;
plot(f1,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 1&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
plot(f2,y,xlim=c(0,4),ylim=c(0,4),main=&amp;quot;model 2&amp;quot;)&lt;br /&gt;
abline(a=0,b=1,lty=1)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Model selection===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Again, ${\cal M}_2$ would seem to have a slight edge. This can be tested more analytically using the [http://en.wikipedia.org/wiki/Bayesian_information_criterion Bayesian Information Criteria] (BIC):&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance1=pk.nlm1$minimum + n*log(2*pi)&lt;br /&gt;
bic1=deviance1+log(n)*length(psi1)&lt;br /&gt;
deviance2=pk.nlm2$minimum + n*log(2*pi)&lt;br /&gt;
bic2=deviance2+log(n)*length(psi2)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
| style=&amp;quot;width:50%&amp;quot; | &lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic1 =&amp;quot;,bic1,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic1 = 24.10972&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic2 =&amp;quot;,bic2,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic2 = 11.24769&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
A smaller BIC is better. Therefore, this also suggests that model ${\cal M}_2$ should be selected.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Fitting different error models===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the moment, we have only considered  constant error models. However, the &amp;quot;observations vs predictions&amp;quot; figure hints that the amplitude of the residual errors may increase with the size of the predicted value. Let us therefore take a closer look at four different residual error models, each of which we will associate with the &amp;quot;best&amp;quot; structural model $f_2$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;2&amp;quot; cellspacing=&amp;quot;8&amp;quot; style=&amp;quot;text-align:left; margin-left:4%&amp;quot;&lt;br /&gt;
|${\cal M}_2$ || Constant error model: || $y_j=f_2(t_j;\phi_2)+a_2\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_3$ || Proportional error model: || $y_j=f_2(t_j;\phi_3)+b_3f_2(t_j;\phi_3)\teps_j$&lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_4$ || Combined error model: || $y_j=f_2(t_j;\phi_4)+(a_4+b_4f_2(t_j;\phi_4))\teps_j$ &lt;br /&gt;
|-&lt;br /&gt;
|${\cal M}_5$ || Exponential error model: || $\log(y_j)=\log(f_2(t_j;\phi_5)) + a_5\teps_j$.&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
The three new ones need to be entered into {{Verbatim|R}}:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
fmin3=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin4=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=abs(x[4])+abs(x[5])*f&lt;br /&gt;
e=sum( ((y-f)/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
fmin5=function(x,y,t)&lt;br /&gt;
{f=predc2(t,x)&lt;br /&gt;
g=x[4]&lt;br /&gt;
e=sum( ((log(y)-log(f))/g)^2 + log(g^2))&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can now compute the MLE $\hatpsi_3=(\hatphi_3,\hat{b}_3)$, $\hatpsi_4=(\hatphi_4,\hat{a}_4,\hat{b}_4)$ and $\hatpsi_5=(\hatphi_5,\hat{a}_5)$ of $\psi$  under models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$:&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;  &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
#----------------  MLE  -------------------&lt;br /&gt;
&lt;br /&gt;
pk.nlm3=nlm(fmin3, c(phi2,0.1), y, t, &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi3=pk.nlm3$estimate&lt;br /&gt;
&lt;br /&gt;
pk.nlm4=nlm(fmin4, c(phi2,1,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi4=pk.nlm4$estimate&lt;br /&gt;
psi4[c(4,5)]=abs(psi4[c(4,5)])&lt;br /&gt;
&lt;br /&gt;
pk.nlm5=nlm(fmin5, c(phi2,0.1), y, t,  &lt;br /&gt;
       hessian=&amp;quot;true&amp;quot;)&lt;br /&gt;
psi5=pk.nlm5$estimate  &lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot; |&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi3 =&amp;quot;,psi3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi3 = 2.642409 11.44113 0.1838779 0.2189221&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi4 =&amp;quot;,psi4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi4 = 2.890066 10.16836 0.2068221 0.02741416 0.1456332&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; psi5 =&amp;quot;,psi5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 psi5 = 2.710984 11.2744 0.188901 0.2310001&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Selecting the error model===&lt;br /&gt;
&lt;br /&gt;
As before, these curves can be plotted over the original data and compared:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:New_Individual4.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
tc=seq(from=0,to=25,by=0.1)&lt;br /&gt;
fc1=predc1(tc,phi1)&lt;br /&gt;
fc2=predc2(tc,phi2)&lt;br /&gt;
&lt;br /&gt;
plot(t,y,ylim=c(0,4.1), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
        ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc1, type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,fc2, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(13,4,c(&amp;quot;observations&amp;quot;, &lt;br /&gt;
        &amp;quot;first order absorption&amp;quot;,&lt;br /&gt;
        &amp;quot;zero order absorption&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,1), pch=c(1,-1,-1), lwd=2, &lt;br /&gt;
        col=c(&amp;quot;blue&amp;quot;,&amp;quot;green&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As you can see, the three predicted concentrations obtained with models ${\cal M}_3$, ${\cal M}_4$  and ${\cal M}_5$ are quite similar. We now calculate the BIC for each:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
deviance3=pk.nlm3$minimum + n*log(2*pi)&lt;br /&gt;
bic3=deviance3 + log(n)*length(psi3)&lt;br /&gt;
deviance4=pk.nlm4$minimum + n*log(2*pi)&lt;br /&gt;
bic4=deviance4 + log(n)*length(psi4)&lt;br /&gt;
deviance5=pk.nlm5$minimum + 2*sum(log(y)) + n*log(2*pi)&lt;br /&gt;
bic5=deviance5 + log(n)*length(psi5)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic3 =&amp;quot;,bic3,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic3 = 3.443607&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic4 =&amp;quot;,bic4,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic4 = 3.475841&lt;br /&gt;
&lt;br /&gt;
&amp;gt; cat(&amp;quot; bic5 =&amp;quot;,bic5,&amp;quot;\n\n&amp;quot;)&lt;br /&gt;
 bic5 = 4.108521&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
All of these BIC are lower than the constant residual error one. BIC selects the residual error model ${\cal M}_3$ with a proportional component.&lt;br /&gt;
&lt;br /&gt;
There is not a large difference between these three error models, though the proportional and combined error models give the smallest and essentially identical BIC.  We decide to use the combined error model ${\cal M}_4$ in the following (the same types of analysis could be done with the proportional error model).&lt;br /&gt;
&lt;br /&gt;
A 90% confidence interval for $\psi_4$ can derived from the Hessian (i.e., the square matrix of second-order partial derivatives)  of the objective function (i.e., -2 $\times \ LL$):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
ialpha=0.9&lt;br /&gt;
df=n-length(phi4)&lt;br /&gt;
I4=pk.nlm4$hessian/2&lt;br /&gt;
H4=solve(I4)&lt;br /&gt;
s4=sqrt(diag(H4)*n/df)&lt;br /&gt;
delta4=s4*qt(0.5+ialpha/2, df)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4&lt;br /&gt;
            [,1]        [,2]&lt;br /&gt;
[1,]  2.22576690  3.55436561&lt;br /&gt;
[2,]  7.93442421 12.40228967&lt;br /&gt;
[3,]  0.16628224  0.24736196&lt;br /&gt;
[4,] -0.02444571  0.07927403&lt;br /&gt;
[5,]  0.04119983  0.25006660&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can also calculate a 90% confidence interval for $f_4(t)$ using the [http://en.wikipedia.org/wiki/Central_limit_theorem Central Limit Theorem] (see [[#intro_individualCLT|(3)]]):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
nlpredci=function(phi,f,H)&lt;br /&gt;
{&lt;br /&gt;
dphi=length(phi)&lt;br /&gt;
nf=length(f)&lt;br /&gt;
H=H*n/(n-dphi)&lt;br /&gt;
S=H[seq(1,dphi),seq(1,dphi)]&lt;br /&gt;
G=matrix(nrow=nf, ncol=dphi)&lt;br /&gt;
for (k in seq(1,dphi)) {&lt;br /&gt;
   dk=phi[k]*(1e-5)&lt;br /&gt;
   phid=phi&lt;br /&gt;
   phid[k]=phi[k] + dk&lt;br /&gt;
   fd=predc2(tc,phid)&lt;br /&gt;
   G[,k]=(f-fd)/dk&lt;br /&gt;
}&lt;br /&gt;
M=rowSums((G%*%S)*G)&lt;br /&gt;
deltaf=sqrt(M)*qt(0.5+ialpha/2,df)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
deltafc4=nlpredci(phi4,fc4,H4)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
This can then be plotted:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual6.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{RcodeForTable&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
plot(t,y,ylim=c(0,4.5), xlab=&amp;quot;time (hour)&amp;quot;, &lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;, col=&amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;,col = &amp;quot;red&amp;quot;,lwd=2)&lt;br /&gt;
lines(tc, fc4-deltafc4, type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot; ,lwd=1, lty=3)&lt;br /&gt;
lines(tc,fc4+deltafc4,type = &amp;quot;l&amp;quot;,&lt;br /&gt;
       col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;,&lt;br /&gt;
       &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;),&lt;br /&gt;
        lty=c(-1,1,3),pch=c(1,-1,-1),lwd=c(2,2,1),&lt;br /&gt;
       col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|} &lt;br /&gt;
&lt;br /&gt;
Alternatively, prediction intervals for $\hatpsi_4$, $\hat{f}_4(t;\hatpsi_4)$ and new observations for any time $t$ can be estimated by Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Rcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
f=predc2(t,phi4)&lt;br /&gt;
a4=psi4[4]&lt;br /&gt;
b4=psi4[5]&lt;br /&gt;
g=a4+b4*f&lt;br /&gt;
dpsi=length(psi4)&lt;br /&gt;
nc=length(tc)&lt;br /&gt;
N=1000&lt;br /&gt;
qalpha=c(0.5 - alpha/2,0.5 + alpha/2)&lt;br /&gt;
PSI=matrix(nrow=N,ncol=dpsi)&lt;br /&gt;
FC=matrix(nrow=N,ncol=nc)&lt;br /&gt;
Y=matrix(nrow=N,ncol=nc)&lt;br /&gt;
for (k in seq(1,N)) {&lt;br /&gt;
   eps=rnorm(n)&lt;br /&gt;
   ys=f+g*eps&lt;br /&gt;
   pk.nlm=nlm(fmin4, psi4, ys, t)&lt;br /&gt;
   psie=pk.nlm$estimate&lt;br /&gt;
   psie[c(4,5)]=abs(psie[c(4,5)])&lt;br /&gt;
   PSI[k,]=psie&lt;br /&gt;
   fce=predc2(tc,psie[c(1,2,3)])&lt;br /&gt;
   FC[k,]=fce&lt;br /&gt;
   gce=a4+b4*fce&lt;br /&gt;
   Y[k,]=fce + gce*rnorm(1)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ci4s=matrix(nrow=dpsi,ncol=2)&lt;br /&gt;
for (k in seq(1,dpsi)){&lt;br /&gt;
   ci4s[k,]=quantile(PSI[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
m4s=colMeans(PSI)&lt;br /&gt;
sd4s=apply(PSI,2,sd)&lt;br /&gt;
&lt;br /&gt;
cifc4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   cifc4s[k,]=quantile(FC[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
ciy4s=matrix(nrow=nc,ncol=2)&lt;br /&gt;
for (k in seq(1,nc)){&lt;br /&gt;
   ciy4s[k,]=quantile(Y[,k],qalpha,names=FALSE)&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
par(mfrow= c(1,1))&lt;br /&gt;
plot(t,y,ylim=c(0,4.5),xlab=&amp;quot;time (hour)&amp;quot;,&lt;br /&gt;
       ylab=&amp;quot;concentration (mg/l)&amp;quot;,col = &amp;quot;blue&amp;quot;)&lt;br /&gt;
lines(tc,fc4, type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=2)&lt;br /&gt;
lines(tc,cifc4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,cifc4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;red&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,1], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
lines(tc,ciy4s[,2], type = &amp;quot;l&amp;quot;, col = &amp;quot;green&amp;quot;, lwd=1, lty=3)&lt;br /&gt;
abline(a=0,b=0,lty=2)&lt;br /&gt;
legend(10.5,4.5,c(&amp;quot;observed concentrations&amp;quot;, &amp;quot;predicted concentration&amp;quot;, &lt;br /&gt;
       &amp;quot;CI for predicted concentration&amp;quot;, &amp;quot;CI for observed concentrations&amp;quot;), &lt;br /&gt;
       lty=c(-1,1,3,3), pch=c(1,-1,-1,-1), lwd=c(2,2,1,1), col=c(&amp;quot;blue&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;red&amp;quot;,&amp;quot;green&amp;quot;))&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;0&amp;quot; &lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:NewIndividual7.png|link=]]&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
{{JustCodeForTable&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none; color:blue&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt; ci4s&lt;br /&gt;
             [,1]        [,2]&lt;br /&gt;
[1,] 2.350653e+00  3.53526320&lt;br /&gt;
[2,] 8.350764e+00 12.04910579&lt;br /&gt;
[3,] 1.818431e-01  0.24156832&lt;br /&gt;
[4,] 5.445459e-09  0.08819339&lt;br /&gt;
[5,] 1.563625e-02  0.19638889&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The R code and input data used in this section can be downloaded here: {{filepath:R_IndividualFitting.rar}}.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{buonaccorsi2010measurement,&lt;br /&gt;
  title={Measurement Error: Models, Methods, and Applications},&lt;br /&gt;
  author={Buonaccorsi, J.P.},&lt;br /&gt;
  isbn={9781420066586},&lt;br /&gt;
  lccn={2009048849},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Interdisciplinary Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=QVtVmaCqLHMC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{carroll2010measurement,&lt;br /&gt;
  title={Measurement Error in Nonlinear Models: A Modern Perspective, Second Edition},&lt;br /&gt;
  author={Carroll, R.J. and Ruppert, D. and Stefanski, L.A. and Crainiceanu, C.M.},&lt;br /&gt;
  isbn={9781420010138},&lt;br /&gt;
  lccn={2006045485},&lt;br /&gt;
  series={Chapman &amp;amp; Hall/CRC Monographs on Statistics &amp;amp; Applied Probability},&lt;br /&gt;
  url={http://books.google.fr/books?id=9kBx5CPZCqkC},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fitzmaurice2004applied,&lt;br /&gt;
  title={Applied Longitudinal Analysis},&lt;br /&gt;
  author={Fitzmaurice, G.M. and Laird, N.M. and Ware, J.H.},&lt;br /&gt;
  isbn={9780471214878},&lt;br /&gt;
  lccn={04040891},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=gCoTIFejMgYC},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{gallant2009nonlinear,&lt;br /&gt;
  title={Nonlinear Statistical Models},&lt;br /&gt;
  author={Gallant, A.R.},&lt;br /&gt;
  isbn={9780470317372},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=imv-NMozseEC},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{huet2003statistical,&lt;br /&gt;
  title={Statistical tools for nonlinear regression: a practical guide with S-PLUS and R examples},&lt;br /&gt;
  author={Huet, S. and Bouvier, A. and Poursat, M.A. and Jolivet, E.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ritz2008nonlinear,&lt;br /&gt;
  title={Nonlinear regression with R},&lt;br /&gt;
  author={Ritz, C. and Streibig, J.C.},&lt;br /&gt;
  volume={33},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ross1990nonlinear,&lt;br /&gt;
  title={Nonlinear estimation},&lt;br /&gt;
  author={Ross, G.J.S.},&lt;br /&gt;
  isbn={9780387972787},&lt;br /&gt;
  lccn={90032797},&lt;br /&gt;
  series={Springer series in statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=7LkyzdLMghIC},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={Springer-Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{seber2003nonlinear,&lt;br /&gt;
  title={Nonlinear Regression},&lt;br /&gt;
  author={Seber, G.A.F. and Wild, C.J.},&lt;br /&gt;
  isbn={9780471471356},&lt;br /&gt;
  lccn={88017194},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=YBYlCpBNo\_cC},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{serroyen2009nonlinear,&lt;br /&gt;
  title={Nonlinear models for longitudinal data},&lt;br /&gt;
  author={Serroyen, J. and Molenberghs, G. and Verbeke, G. and Davidian, M. },&lt;br /&gt;
  journal={The American Statistician},&lt;br /&gt;
  volume={63},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={378-388},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wolberg2006data,&lt;br /&gt;
  title={Data analysis using the method of least squares: extracting the most information from experiments},&lt;br /&gt;
  author={Wolberg, J.R.},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Springer Berlin, Germany}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Overview &lt;br /&gt;
|linkNext=What is a model? A joint probability distribution! }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7466</id>
		<title>Estimation of the observed Fisher information matrix</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7466"/>
		<updated>2013-08-28T10:09:02Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;$ \def\hphi{\tilde{\phi}} $&lt;br /&gt;
==Estimation using stochastic approximation==&lt;br /&gt;
&lt;br /&gt;
The ''observed'' Fisher information matrix (F.I.M.) is a function of $\theta$ defined as&lt;br /&gt;
  &lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq_fim1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
I(\theta) &amp;amp;=&amp;amp; -\DDt{\log ({\like}(\theta;\by))} \\&lt;br /&gt;
&amp;amp;=&amp;amp; -\DDt{\log (\py(\by;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
Due to the likelihood being quite complex, $I(\theta)$ usually has no closed form expression. It is however possible to estimate it using a stochastic approximation procedure based on &amp;lt;balloon title=&amp;quot;Kuhn05: put here the reference!!!&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;Louis' formula&amp;lt;/balloon&amp;gt;:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\DDt{\log (\pmacro(\by;\theta))} = \esp{\DDt{\log (\pmacro(\by,\bpsi;\theta))} {{!}}  \by ;\theta} + \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta},&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
 \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta} &amp;amp;=&amp;amp;&lt;br /&gt;
  \esp{ \left(\Dt{\log (\pmacro(\by,\bpsi;\theta))} \right)\left(\Dt{\log (\pmacro(\by,\bpsi;\theta))}\right)^{\transpose} {{!}} \by ; \theta} \\&lt;br /&gt;
&amp;amp;&amp;amp; - \esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta}\esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta}^{\transpose} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus, $\DDt{\log (\pmacro(\by;\theta))}$  is defined as a combination of conditional expectations. Each of these conditional expectations can be estimated by Monte Carlo, or equivalently approximated using a stochastic approximation algorithm.&lt;br /&gt;
&lt;br /&gt;
We can then draw a sequence  $(\psi_i^{(k)})$ using a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hasting algorithm]] and estimate the observed F.I.M. online. At iteration $k$ of the algorithm:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Simulation step''': for $i=1,2,\ldots,N$, draw $\psi_i^{(k)}$ from $m$ iterations of the Metropolis-Hastings algorithm described in [[The Metropolis-Hastings algorithm for simulating the individual parameters| The Metropolis-Hastings algorithm]] section with $\pmacro(\psi_i |y_i ;{\theta})$ as the limit distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Stochastic approximation''': update $D_k$, $G_k$ and $\Delta_k$  according to the following recurrence relations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k &amp;amp; = &amp;amp; \Delta_{k-1} + \gamma_k \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - \Delta_{k-1} \right) \\&lt;br /&gt;
D_k &amp;amp; = &amp;amp; D_{k-1} + \gamma_k \left(\DDt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - D_{k-1} \right)\\&lt;br /&gt;
G_k &amp;amp; = &amp;amp; G_{k-1} + \gamma_k \left((\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})^\transpose -G_{k-1} \right),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $(\gamma_k)$ is a decreasing sequence of positive numbers such that $\gamma_1=1$, $ \sum_{k=1}^{\infty} \gamma_k = \infty$,  and $\sum_{k=1}^{\infty} \gamma_k^2 &amp;lt; \infty$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Estimation step''': update the estimate $H_k$ of the F.I.M. according to&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;H_k  =  D_k + G_k - \Delta_k \Delta_k^{\transpose}. &amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Implementing this algorithm therefore requires computation of the first and second derivatives of&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log (\pmacro(\by,\bpsi;\theta))=\sum_{i=1}^{N} \log (\pmacro(y_i,\psi_i;\theta)).&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Assume first that the joint distribution of $\by$ and $\bpsi$ decomposes as&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_dec1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pypsi(\by,\bpsi;\theta) = \pcypsi(\by {{!}} \bpsi)\ppsi(\bpsi;\theta).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
This assumption means that for any $i=1,2,\ldots,N$, all of the components of $\psi_i$ are random and  there exists a sufficient statistic ${\cal S}(\bpsi)$ for the estimation of $\theta$. It is then sufficient to compute the first and second derivatives of $\log (\pmacro(\bpsi;\theta))$ in order to estimate the F.I.M. This can be done relatively simply in closed form when the individual parameters are normally distributed (or a transformation $h$ of them is).&lt;br /&gt;
&lt;br /&gt;
If some component of $\psi_i$ has no variability, [[#eq:fim_dec1|(2)]] no longer holds, but we can decompose $\theta$ into $(\theta_y,\theta_\psi)$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) = \pcyipsii(y_i {{!}} \psi_i ; \theta_y)\ppsii(\psi_i;\theta_\psi).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We  then need to compute the first and second derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ and $\log(\ppsii(\psi_i;\theta_\psi))$. Derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ that do not have a closed form expression can be obtained using central differences.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
1. Using $\gamma_k=1/k$ for $k \geq 1$ means that each term is approximated with an empirical mean obtained from $(\bpsi^{(k)}, k \geq 1)$. For instance,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k&lt;br /&gt;
&amp;amp;=&amp;amp;  \Delta_{k-1} + \displaystyle{ \frac{1}{k} } \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};\theta))} - \Delta_{k-1} \right)   &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
&amp;amp;=&amp;amp;  \displaystyle{ \frac{1}{k} }\sum_{j=1}^{k} \Dt{\log (\pmacro(\by,\bpsi^{(j)};\theta))} .  &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
[[#eq:fim_Delta1|(3)]] (resp. [[#eq:fim_Delta2|(4)]]) defines $\Delta_k$ using an online (resp. offline) algorithm. Writing $\Delta_k$ as in [[#eq:fim_Delta1|(3)]] instead of [[#eq:fim_Delta2|(4)]] avoids having to store all simulated sequences $(\bpsi^{(j)}, 1\leq j \leq k)$ when computing $\Delta_k$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2. This approach is used  for computing the F.I.M.  $I(\hat{\theta})$ in practice, where $\hat{\theta}$ is the maximum likelihood estimate of $\theta$. The only difference with the [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings]] used for SAEM is that the population parameter $\theta$ is not updated and remains fixed at  $\hat{\theta}$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, a stochastic approximation algorithm for estimating the observed Fisher Information Matrix $I(\hat{\theta)}$ consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, run a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings algorithm]] to draw a sequence $\psi_i^{(k)}$ with limit distribution $\pmacro(\psi_i {{!}}y_i ;\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. At iteration $k$ of the Metropolis-Hastings algorithm, compute the first and second derivatives of $\pypsi(\by,\bpsi^{(k)};\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3.Update $\Delta_k$, $G_k$, $D_k$ and compute an estimate $H_k$ of the F.I.M.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 1&lt;br /&gt;
|text=Consider the model&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_i {{!}} \psi_i &amp;amp;\sim&amp;amp; \pcyipsii(y_i {{!}} \psi_i) \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\Omega = {\rm diag}(\omega_1^2,\omega_2^2,\ldots,\omega_d^2)$ is a diagonal matrix and $h(\psi_i)=(h_1(\psi_{i,1}), h_2(\psi_{i,2}), \ldots , h_d(\psi_{i,d}) )^{\transpose}$.&lt;br /&gt;
The vector of population parameters is $\theta = (\psi_{\rm pop} , \Omega)=(\psi_{ {\rm pop},1},\ldots,\psi_{ {\rm pop},d},\omega_1^2,\ldots,\omega_d^2)$.&lt;br /&gt;
&lt;br /&gt;
Here,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log (\pyipsii(y_i,\psi_i;\theta)) = \log (\pcyipsii(y_i {{!}} \psi_i)) + \log (\ppsii(\psi_i;\theta)).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Dt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \Dt{\log (\ppsii(\psi_i;\theta))} \\&lt;br /&gt;
\DDt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \DDt{\log (\ppsii(\psi_i;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
More precisely,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\log (\ppsii(\psi_i;\theta)) &amp;amp;=&amp;amp; -\displaystyle{\frac{d}{2} }\log(2\pi) + \sum_{\iparam=1}^d \log(h_\iparam^{\prime}(\psi_{i,\iparam}))&lt;br /&gt;
-\displaystyle{ \frac{1}{2} } \sum_{\iparam=1}^d \log(\omega_\iparam^2)&lt;br /&gt;
-\sum_{\iparam=1}^d \displaystyle{ \frac{1}{2\, \omega_\iparam^2} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
\displaystyle{\frac{1}{\omega_\iparam^2} }h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) ) \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
-\displaystyle{ \frac{1}{2\omega_\iparam^2} }&lt;br /&gt;
+\displaystyle{\frac{1}{2\, \omega_\iparam^4} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \psi_{ {\rm pop},\jparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
 \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{\omega_\iparam^2} --&amp;gt;&lt;br /&gt;
\left( h_\iparam^{\prime\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )- h_\iparam^{\prime \, 2}(\psi_{ {\rm pop},\iparam}) \right)/\omega_\iparam^2 &amp;amp; {\rm if \quad } \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{2\omega_\iparam^4} - \frac{1}{\omega_\iparam^6} --&amp;gt;&lt;br /&gt;
1/(2\omega_\iparam^4) -&lt;br /&gt;
( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2/\omega_\iparam^6 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %      -\frac{1}{\omega_\iparam^4} --&amp;gt;&lt;br /&gt;
-h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )/\omega_\iparam^4 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 2&lt;br /&gt;
|text= We consider the same  model for continuous data, assuming a constant error model and  that the variance $a^2$ of the residual error has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $\theta_y=a^2$, $\theta_\psi=(\psi_{\rm pop},\Omega)$ and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pyipsii(y_i,\psi_i;\theta)) = \log(\pcyipsii(y_i {{!}} \psi_i ; a^2)) + \log(\ppsii(\psi_i;\psi_{\rm pop},\Omega)),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))$ with respect to $a^2$ are straightforward to compute. Derivatives of $\log(\ppsii(\psi_i;\psi_{\rm pop},\Omega))$ with respect to $\psi_{\rm pop}$ and $\Omega$ remain unchanged.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 3&lt;br /&gt;
|text=  Consider again the same  model for continuous data,  assuming now that a subset $\xi$ of the parameters of the structural model has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i,\xi) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\psi$ remain as the subset of individual parameters with variability. Here, $\theta_y=(\xi,a^2)$, $\theta_\psi=(\psi_{\rm pop},\Omega)$, and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; \xi,a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2 a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i,\xi))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; \xi, a^2))$ with respect to $\xi$ require  computation of the derivative of $f$ with respect to $\xi$. These derivatives are usually not calculable. One possibility is to numerically approximate them using finite differences.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation using linearization of the model == &lt;br /&gt;
&lt;br /&gt;
Consider here a model for continuous data that uses a $\phi$-parametrization for the individual parameters:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;= &amp;amp; f(t_{ij} , \phi_i) + g(t_{ij} , \phi_i)\teps_{ij} \\&lt;br /&gt;
\phi_i &amp;amp;=&amp;amp; \phi_{\rm pop} + \eta_i .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\hphi_i$ be some predicted value of $\phi_i$, such as for instance the estimated mean or  estimated mode of the conditional distribution $\pmacro(\phi_i |y_i ; \hat{\theta})$.&lt;br /&gt;
&lt;br /&gt;
We can then choose to linearize the model for the observations $(y_{ij}, 1\leq j \leq n_i)$ of individual $i$ around the vector of predicted individual parameters. Let $\Dphi{f(t , \phi)}$ be the row vector of derivatives of  $f(t , \phi)$ with respect to $\phi$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) + \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_i - \hphi_i)  + g(t_{ij} , \hphi_i)\teps_{ij} \\&lt;br /&gt;
&amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) +  \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i)&lt;br /&gt;
+  \Dphi{f(t_{ij} , \hphi_i)} \, \eta_i  + g(t_{ij} , \hphi_i)\teps_{ij} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, we can approximate the marginal distribution of the vector $y_i$ as a normal distribution:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_approx&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{i} \approx {\cal N}\left(f(t_{i} , \hphi_i) +  \Dphi{f(t_{i} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i) ,&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} \Omega \Dphi{f(t_{i} , \hphi_i)}^{\transpose}  + g(t_{i} , \hphi_i)\Sigma_{n_i} g(t_{ij} , \hphi_i)^{\transpose} \right),&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&lt;br /&gt;
where $\Sigma_{n_i}$ is the variance-covariance matrix of $\teps_{i,1},\ldots,\teps_{i,n_i}$. If the $\teps_{ij}$ are i.i.d., then&lt;br /&gt;
$\Sigma_{n_i}$ is the identity matrix.&lt;br /&gt;
&lt;br /&gt;
We can equivalently use the original $\psi$-parametrization and the fact that $\phi_i=h(\psi_i)$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} = \Dpsi{f(t_{i} , \hpsi_i)} J_h(\hpsi_i)^{\transpose} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $J_h$ is the Jacobian of $h$.&lt;br /&gt;
&lt;br /&gt;
We then can approximate the observed log-likelihood ${\llike}(\theta) = \log(\like(\theta;\by))=\sum_{i=1}^N \log(\pyi(y_i;\theta))$ using this normal approximation. We can also derive the F.I.M. by computing the matrix of second-order partial derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&lt;br /&gt;
Except for very simple models, computing these second-order partial derivatives in  closed form is not straightforward. In such cases, finite differences can be used for numerically approximating  them. We can use for instance a central difference approximation of the second derivative of $\llike(\theta)$. To this end, let $\nu&amp;gt;0$. For $j=1,2,\ldots, m$, let $\nu^{(j)}=(\nu^{(j)}_{k}, 1\leq k \leq m)$ be the $m$-vector such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\nu^{(j)}_{k} = \left\{&lt;br /&gt;
                  \begin{array}{ll}&lt;br /&gt;
                    \nu &amp;amp; {\rm if \quad j= k} \\&lt;br /&gt;
                    0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
                  \end{array}&lt;br /&gt;
                \right.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, for $\nu$ small enough,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial_{\theta_j}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{ \frac{ {\llike}(\theta+\nu^{(j)})- {\llike}(\theta-\nu^{(j)})}{2\nu} }  \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_diff&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial^2_{\theta_j,\theta_k}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{\frac{ {\llike}(\theta+\nu^{(j)}+\nu^{(k)})- {\llike}(\theta+\nu^{(j)}-\nu^{(k)})&lt;br /&gt;
-{\llike}(\theta-\nu^{(j)}+\nu^{(k)})+{\llike}(\theta-\nu^{(j)}-\nu^{(k)})}{4\nu^2} } . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, the algorithm for approximating the Fisher Information Matrix $I(\hat{\theta)}$ using a linear approximation of the model consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, obtain some estimate $(\hpsi_i)$ of the individual parameters $(\psi_i)$ (we can average for example the final terms of the sequence $(\psi_i^{(k)})$ drawn during the final iterations of the [[The SAEM algorithm for estimating population parameters| SAEM algorithm]]).&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. For $i=1,2,\ldots,N$, compute $\hphi_i=h(\hpsi_i)$,  the mean and the variance of the normal distribution defined in [[#eq:fim_approx|(5)]], and  ${\llike}(\theta)$ using this normal approximation.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3. Use  [[#eq:fim_diff|(6)]] to approximate the matrix of second-order derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkNext=Estimation of the log-likelihood&lt;br /&gt;
|linkBack=The Metropolis-Hastings algorithm for simulating the individual parameters }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7465</id>
		<title>Estimation of the observed Fisher information matrix</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7465"/>
		<updated>2013-08-28T10:07:29Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Estimation using stochastic approximation */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;==Estimation using stochastic approximation==&lt;br /&gt;
&lt;br /&gt;
The ''observed'' Fisher information matrix (F.I.M.) is a function of $\theta$ defined as&lt;br /&gt;
  &lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq_fim1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
I(\theta) &amp;amp;=&amp;amp; -\DDt{\log ({\like}(\theta;\by))} \\&lt;br /&gt;
&amp;amp;=&amp;amp; -\DDt{\log (\py(\by;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
Due to the likelihood being quite complex, $I(\theta)$ usually has no closed form expression. It is however possible to estimate it using a stochastic approximation procedure based on &amp;lt;balloon title=&amp;quot;Kuhn05: put here the reference!!!&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;Louis' formula&amp;lt;/balloon&amp;gt;:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\DDt{\log (\pmacro(\by;\theta))} = \esp{\DDt{\log (\pmacro(\by,\bpsi;\theta))} {{!}}  \by ;\theta} + \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta},&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
 \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta} &amp;amp;=&amp;amp;&lt;br /&gt;
  \esp{ \left(\Dt{\log (\pmacro(\by,\bpsi;\theta))} \right)\left(\Dt{\log (\pmacro(\by,\bpsi;\theta))}\right)^{\transpose} {{!}} \by ; \theta} \\&lt;br /&gt;
&amp;amp;&amp;amp; - \esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta}\esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} {{!}} \by ; \theta}^{\transpose} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus, $\DDt{\log (\pmacro(\by;\theta))}$  is defined as a combination of conditional expectations. Each of these conditional expectations can be estimated by Monte Carlo, or equivalently approximated using a stochastic approximation algorithm.&lt;br /&gt;
&lt;br /&gt;
We can then draw a sequence  $(\psi_i^{(k)})$ using a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hasting algorithm]] and estimate the observed F.I.M. online. At iteration $k$ of the algorithm:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Simulation step''': for $i=1,2,\ldots,N$, draw $\psi_i^{(k)}$ from $m$ iterations of the Metropolis-Hastings algorithm described in [[The Metropolis-Hastings algorithm for simulating the individual parameters| The Metropolis-Hastings algorithm]] section with $\pmacro(\psi_i |y_i ;{\theta})$ as the limit distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Stochastic approximation''': update $D_k$, $G_k$ and $\Delta_k$  according to the following recurrence relations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k &amp;amp; = &amp;amp; \Delta_{k-1} + \gamma_k \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - \Delta_{k-1} \right) \\&lt;br /&gt;
D_k &amp;amp; = &amp;amp; D_{k-1} + \gamma_k \left(\DDt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - D_{k-1} \right)\\&lt;br /&gt;
G_k &amp;amp; = &amp;amp; G_{k-1} + \gamma_k \left((\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})^\transpose -G_{k-1} \right),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $(\gamma_k)$ is a decreasing sequence of positive numbers such that $\gamma_1=1$, $ \sum_{k=1}^{\infty} \gamma_k = \infty$,  and $\sum_{k=1}^{\infty} \gamma_k^2 &amp;lt; \infty$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Estimation step''': update the estimate $H_k$ of the F.I.M. according to&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;H_k  =  D_k + G_k - \Delta_k \Delta_k^{\transpose}. &amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Implementing this algorithm therefore requires computation of the first and second derivatives of&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log (\pmacro(\by,\bpsi;\theta))=\sum_{i=1}^{N} \log (\pmacro(y_i,\psi_i;\theta)).&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Assume first that the joint distribution of $\by$ and $\bpsi$ decomposes as&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_dec1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pypsi(\by,\bpsi;\theta) = \pcypsi(\by {{!}} \bpsi)\ppsi(\bpsi;\theta).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
This assumption means that for any $i=1,2,\ldots,N$, all of the components of $\psi_i$ are random and  there exists a sufficient statistic ${\cal S}(\bpsi)$ for the estimation of $\theta$. It is then sufficient to compute the first and second derivatives of $\log (\pmacro(\bpsi;\theta))$ in order to estimate the F.I.M. This can be done relatively simply in closed form when the individual parameters are normally distributed (or a transformation $h$ of them is).&lt;br /&gt;
&lt;br /&gt;
If some component of $\psi_i$ has no variability, [[#eq:fim_dec1|(2)]] no longer holds, but we can decompose $\theta$ into $(\theta_y,\theta_\psi)$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) = \pcyipsii(y_i {{!}} \psi_i ; \theta_y)\ppsii(\psi_i;\theta_\psi).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We  then need to compute the first and second derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ and $\log(\ppsii(\psi_i;\theta_\psi))$. Derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ that do not have a closed form expression can be obtained using central differences.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
1. Using $\gamma_k=1/k$ for $k \geq 1$ means that each term is approximated with an empirical mean obtained from $(\bpsi^{(k)}, k \geq 1)$. For instance,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k&lt;br /&gt;
&amp;amp;=&amp;amp;  \Delta_{k-1} + \displaystyle{ \frac{1}{k} } \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};\theta))} - \Delta_{k-1} \right)   &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
&amp;amp;=&amp;amp;  \displaystyle{ \frac{1}{k} }\sum_{j=1}^{k} \Dt{\log (\pmacro(\by,\bpsi^{(j)};\theta))} .  &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
[[#eq:fim_Delta1|(3)]] (resp. [[#eq:fim_Delta2|(4)]]) defines $\Delta_k$ using an online (resp. offline) algorithm. Writing $\Delta_k$ as in [[#eq:fim_Delta1|(3)]] instead of [[#eq:fim_Delta2|(4)]] avoids having to store all simulated sequences $(\bpsi^{(j)}, 1\leq j \leq k)$ when computing $\Delta_k$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2. This approach is used  for computing the F.I.M.  $I(\hat{\theta})$ in practice, where $\hat{\theta}$ is the maximum likelihood estimate of $\theta$. The only difference with the [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings]] used for SAEM is that the population parameter $\theta$ is not updated and remains fixed at  $\hat{\theta}$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, a stochastic approximation algorithm for estimating the observed Fisher Information Matrix $I(\hat{\theta)}$ consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, run a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings algorithm]] to draw a sequence $\psi_i^{(k)}$ with limit distribution $\pmacro(\psi_i {{!}}y_i ;\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. At iteration $k$ of the Metropolis-Hastings algorithm, compute the first and second derivatives of $\pypsi(\by,\bpsi^{(k)};\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3.Update $\Delta_k$, $G_k$, $D_k$ and compute an estimate $H_k$ of the F.I.M.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 1&lt;br /&gt;
|text=Consider the model&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_i {{!}} \psi_i &amp;amp;\sim&amp;amp; \pcyipsii(y_i {{!}} \psi_i) \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\Omega = {\rm diag}(\omega_1^2,\omega_2^2,\ldots,\omega_d^2)$ is a diagonal matrix and $h(\psi_i)=(h_1(\psi_{i,1}), h_2(\psi_{i,2}), \ldots , h_d(\psi_{i,d}) )^{\transpose}$.&lt;br /&gt;
The vector of population parameters is $\theta = (\psi_{\rm pop} , \Omega)=(\psi_{ {\rm pop},1},\ldots,\psi_{ {\rm pop},d},\omega_1^2,\ldots,\omega_d^2)$.&lt;br /&gt;
&lt;br /&gt;
Here,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log (\pyipsii(y_i,\psi_i;\theta)) = \log (\pcyipsii(y_i {{!}} \psi_i)) + \log (\ppsii(\psi_i;\theta)).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Dt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \Dt{\log (\ppsii(\psi_i;\theta))} \\&lt;br /&gt;
\DDt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \DDt{\log (\ppsii(\psi_i;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
More precisely,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\log (\ppsii(\psi_i;\theta)) &amp;amp;=&amp;amp; -\displaystyle{\frac{d}{2} }\log(2\pi) + \sum_{\iparam=1}^d \log(h_\iparam^{\prime}(\psi_{i,\iparam}))&lt;br /&gt;
-\displaystyle{ \frac{1}{2} } \sum_{\iparam=1}^d \log(\omega_\iparam^2)&lt;br /&gt;
-\sum_{\iparam=1}^d \displaystyle{ \frac{1}{2\, \omega_\iparam^2} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
\displaystyle{\frac{1}{\omega_\iparam^2} }h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) ) \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
-\displaystyle{ \frac{1}{2\omega_\iparam^2} }&lt;br /&gt;
+\displaystyle{\frac{1}{2\, \omega_\iparam^4} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \psi_{ {\rm pop},\jparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
 \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{\omega_\iparam^2} --&amp;gt;&lt;br /&gt;
\left( h_\iparam^{\prime\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )- h_\iparam^{\prime \, 2}(\psi_{ {\rm pop},\iparam}) \right)/\omega_\iparam^2 &amp;amp; {\rm if \quad } \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{2\omega_\iparam^4} - \frac{1}{\omega_\iparam^6} --&amp;gt;&lt;br /&gt;
1/(2\omega_\iparam^4) -&lt;br /&gt;
( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2/\omega_\iparam^6 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %      -\frac{1}{\omega_\iparam^4} --&amp;gt;&lt;br /&gt;
-h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )/\omega_\iparam^4 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 2&lt;br /&gt;
|text= We consider the same  model for continuous data, assuming a constant error model and  that the variance $a^2$ of the residual error has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $\theta_y=a^2$, $\theta_\psi=(\psi_{\rm pop},\Omega)$ and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pyipsii(y_i,\psi_i;\theta)) = \log(\pcyipsii(y_i {{!}} \psi_i ; a^2)) + \log(\ppsii(\psi_i;\psi_{\rm pop},\Omega)),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))$ with respect to $a^2$ are straightforward to compute. Derivatives of $\log(\ppsii(\psi_i;\psi_{\rm pop},\Omega))$ with respect to $\psi_{\rm pop}$ and $\Omega$ remain unchanged.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 3&lt;br /&gt;
|text=  Consider again the same  model for continuous data,  assuming now that a subset $\xi$ of the parameters of the structural model has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i,\xi) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\psi$ remain as the subset of individual parameters with variability. Here, $\theta_y=(\xi,a^2)$, $\theta_\psi=(\psi_{\rm pop},\Omega)$, and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; \xi,a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2 a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i,\xi))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; \xi, a^2))$ with respect to $\xi$ require  computation of the derivative of $f$ with respect to $\xi$. These derivatives are usually not calculable. One possibility is to numerically approximate them using finite differences.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation using linearization of the model == &lt;br /&gt;
&lt;br /&gt;
Consider here a model for continuous data that uses a $\phi$-parametrization for the individual parameters:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;= &amp;amp; f(t_{ij} , \phi_i) + g(t_{ij} , \phi_i)\teps_{ij} \\&lt;br /&gt;
\phi_i &amp;amp;=&amp;amp; \phi_{\rm pop} + \eta_i .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\hphi_i$ be some predicted value of $\phi_i$, such as for instance the estimated mean or  estimated mode of the conditional distribution $\pmacro(\phi_i |y_i ; \hat{\theta})$.&lt;br /&gt;
&lt;br /&gt;
We can then choose to linearize the model for the observations $(y_{ij}, 1\leq j \leq n_i)$ of individual $i$ around the vector of predicted individual parameters. Let $\Dphi{f(t , \phi)}$ be the row vector of derivatives of  $f(t , \phi)$ with respect to $\phi$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) + \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_i - \hphi_i)  + g(t_{ij} , \hphi_i)\teps_{ij} \\&lt;br /&gt;
&amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) +  \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i)&lt;br /&gt;
+  \Dphi{f(t_{ij} , \hphi_i)} \, \eta_i  + g(t_{ij} , \hphi_i)\teps_{ij} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, we can approximate the marginal distribution of the vector $y_i$ as a normal distribution:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_approx&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{i} \approx {\cal N}\left(f(t_{i} , \hphi_i) +  \Dphi{f(t_{i} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i) ,&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} \Omega \Dphi{f(t_{i} , \hphi_i)}^{\transpose}  + g(t_{i} , \hphi_i)\Sigma_{n_i} g(t_{ij} , \hphi_i)^{\transpose} \right),&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&lt;br /&gt;
where $\Sigma_{n_i}$ is the variance-covariance matrix of $\teps_{i,1},\ldots,\teps_{i,n_i}$. If the $\teps_{ij}$ are i.i.d., then&lt;br /&gt;
$\Sigma_{n_i}$ is the identity matrix.&lt;br /&gt;
&lt;br /&gt;
We can equivalently use the original $\psi$-parametrization and the fact that $\phi_i=h(\psi_i)$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} = \Dpsi{f(t_{i} , \hpsi_i)} J_h(\hpsi_i)^{\transpose} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $J_h$ is the Jacobian of $h$.&lt;br /&gt;
&lt;br /&gt;
We then can approximate the observed log-likelihood ${\llike}(\theta) = \log(\like(\theta;\by))=\sum_{i=1}^N \log(\pyi(y_i;\theta))$ using this normal approximation. We can also derive the F.I.M. by computing the matrix of second-order partial derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&lt;br /&gt;
Except for very simple models, computing these second-order partial derivatives in  closed form is not straightforward. In such cases, finite differences can be used for numerically approximating  them. We can use for instance a central difference approximation of the second derivative of $\llike(\theta)$. To this end, let $\nu&amp;gt;0$. For $j=1,2,\ldots, m$, let $\nu^{(j)}=(\nu^{(j)}_{k}, 1\leq k \leq m)$ be the $m$-vector such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\nu^{(j)}_{k} = \left\{&lt;br /&gt;
                  \begin{array}{ll}&lt;br /&gt;
                    \nu &amp;amp; {\rm if \quad j= k} \\&lt;br /&gt;
                    0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
                  \end{array}&lt;br /&gt;
                \right.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, for $\nu$ small enough,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial_{\theta_j}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{ \frac{ {\llike}(\theta+\nu^{(j)})- {\llike}(\theta-\nu^{(j)})}{2\nu} }  \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_diff&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial^2_{\theta_j,\theta_k}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{\frac{ {\llike}(\theta+\nu^{(j)}+\nu^{(k)})- {\llike}(\theta+\nu^{(j)}-\nu^{(k)})&lt;br /&gt;
-{\llike}(\theta-\nu^{(j)}+\nu^{(k)})+{\llike}(\theta-\nu^{(j)}-\nu^{(k)})}{4\nu^2} } . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, the algorithm for approximating the Fisher Information Matrix $I(\hat{\theta)}$ using a linear approximation of the model consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, obtain some estimate $(\hpsi_i)$ of the individual parameters $(\psi_i)$ (we can average for example the final terms of the sequence $(\psi_i^{(k)})$ drawn during the final iterations of the [[The SAEM algorithm for estimating population parameters| SAEM algorithm]]).&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. For $i=1,2,\ldots,N$, compute $\hphi_i=h(\hpsi_i)$,  the mean and the variance of the normal distribution defined in [[#eq:fim_approx|(5)]], and  ${\llike}(\theta)$ using this normal approximation.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3. Use  [[#eq:fim_diff|(6)]] to approximate the matrix of second-order derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkNext=Estimation of the log-likelihood&lt;br /&gt;
|linkBack=The Metropolis-Hastings algorithm for simulating the individual parameters }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7464</id>
		<title>Estimation of the observed Fisher information matrix</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7464"/>
		<updated>2013-08-28T10:05:55Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Estimation using stochastic approximation */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;==Estimation using stochastic approximation==&lt;br /&gt;
&lt;br /&gt;
The ''observed'' Fisher information matrix (F.I.M.) is a function of $\theta$ defined as&lt;br /&gt;
  &lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq_fim1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
I(\theta) &amp;amp;=&amp;amp; -\DDt{\log ({\like}(\theta;\by))} \\&lt;br /&gt;
&amp;amp;=&amp;amp; -\DDt{\log (\py(\by;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
Due to the likelihood being quite complex, $I(\theta)$ usually has no closed form expression. It is however possible to estimate it using a stochastic approximation procedure based on &amp;lt;balloon title=&amp;quot;Kuhn05: put here the reference!!!&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;Louis' formula&amp;lt;/balloon&amp;gt;:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\DDt{\log (\pmacro(\by;\theta))} = \esp{\DDt{\log (\pmacro(\by,\bpsi;\theta))}| \by ;\theta} + \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta},&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
 \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta} &amp;amp;=&amp;amp;&lt;br /&gt;
  \esp{ \left(\Dt{\log (\pmacro(\by,\bpsi;\theta))} \right)\left(\Dt{\log (\pmacro(\by,\bpsi;\theta))}\right)^{\transpose} | \by ; \theta} \\&lt;br /&gt;
&amp;amp;&amp;amp; - \esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta}\esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta}^{\transpose} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus, $\DDt{\log (\pmacro(\by;\theta))}$  is defined as a combination of conditional expectations. Each of these conditional expectations can be estimated by Monte Carlo, or equivalently approximated using a stochastic approximation algorithm.&lt;br /&gt;
&lt;br /&gt;
We can then draw a sequence  $(\psi_i^{(k)})$ using a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hasting algorithm]] and estimate the observed F.I.M. online. At iteration $k$ of the algorithm:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Simulation step''': for $i=1,2,\ldots,N$, draw $\psi_i^{(k)}$ from $m$ iterations of the Metropolis-Hastings algorithm described in [[The Metropolis-Hastings algorithm for simulating the individual parameters| The Metropolis-Hastings algorithm]] section with $\pmacro(\psi_i |y_i ;{\theta})$ as the limit distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Stochastic approximation''': update $D_k$, $G_k$ and $\Delta_k$  according to the following recurrence relations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k &amp;amp; = &amp;amp; \Delta_{k-1} + \gamma_k \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - \Delta_{k-1} \right) \\&lt;br /&gt;
D_k &amp;amp; = &amp;amp; D_{k-1} + \gamma_k \left(\DDt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - D_{k-1} \right)\\&lt;br /&gt;
G_k &amp;amp; = &amp;amp; G_{k-1} + \gamma_k \left((\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})^\transpose -G_{k-1} \right),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $(\gamma_k)$ is a decreasing sequence of positive numbers such that $\gamma_1=1$, $ \sum_{k=1}^{\infty} \gamma_k = \infty$,  and $\sum_{k=1}^{\infty} \gamma_k^2 &amp;lt; \infty$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Estimation step''': update the estimate $H_k$ of the F.I.M. according to&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;H_k  =  D_k + G_k - \Delta_k \Delta_k^{\transpose}. &amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Implementing this algorithm therefore requires computation of the first and second derivatives of&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log (\pmacro(\by,\bpsi;\theta))=\sum_{i=1}^{N} \log (\pmacro(y_i,\psi_i;\theta)).&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Assume first that the joint distribution of $\by$ and $\bpsi$ decomposes as&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_dec1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pypsi(\by,\bpsi;\theta) = \pcypsi(\by {{!}} \bpsi)\ppsi(\bpsi;\theta).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
This assumption means that for any $i=1,2,\ldots,N$, all of the components of $\psi_i$ are random and  there exists a sufficient statistic ${\cal S}(\bpsi)$ for the estimation of $\theta$. It is then sufficient to compute the first and second derivatives of $\log (\pmacro(\bpsi;\theta))$ in order to estimate the F.I.M. This can be done relatively simply in closed form when the individual parameters are normally distributed (or a transformation $h$ of them is).&lt;br /&gt;
&lt;br /&gt;
If some component of $\psi_i$ has no variability, [[#eq:fim_dec1|(2)]] no longer holds, but we can decompose $\theta$ into $(\theta_y,\theta_\psi)$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) = \pcyipsii(y_i {{!}} \psi_i ; \theta_y)\ppsii(\psi_i;\theta_\psi).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We  then need to compute the first and second derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ and $\log(\ppsii(\psi_i;\theta_\psi))$. Derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ that do not have a closed form expression can be obtained using central differences.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
1. Using $\gamma_k=1/k$ for $k \geq 1$ means that each term is approximated with an empirical mean obtained from $(\bpsi^{(k)}, k \geq 1)$. For instance,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k&lt;br /&gt;
&amp;amp;=&amp;amp;  \Delta_{k-1} + \displaystyle{ \frac{1}{k} } \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};\theta))} - \Delta_{k-1} \right)   &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
&amp;amp;=&amp;amp;  \displaystyle{ \frac{1}{k} }\sum_{j=1}^{k} \Dt{\log (\pmacro(\by,\bpsi^{(j)};\theta))} .  &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
[[#eq:fim_Delta1|(3)]] (resp. [[#eq:fim_Delta2|(4)]]) defines $\Delta_k$ using an online (resp. offline) algorithm. Writing $\Delta_k$ as in [[#eq:fim_Delta1|(3)]] instead of [[#eq:fim_Delta2|(4)]] avoids having to store all simulated sequences $(\bpsi^{(j)}, 1\leq j \leq k)$ when computing $\Delta_k$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2. This approach is used  for computing the F.I.M.  $I(\hat{\theta})$ in practice, where $\hat{\theta}$ is the maximum likelihood estimate of $\theta$. The only difference with the [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings]] used for SAEM is that the population parameter $\theta$ is not updated and remains fixed at  $\hat{\theta}$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, a stochastic approximation algorithm for estimating the observed Fisher Information Matrix $I(\hat{\theta)}$ consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, run a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings algorithm]] to draw a sequence $\psi_i^{(k)}$ with limit distribution $\pmacro(\psi_i {{!}}y_i ;\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. At iteration $k$ of the Metropolis-Hastings algorithm, compute the first and second derivatives of $\pypsi(\by,\bpsi^{(k)};\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3.Update $\Delta_k$, $G_k$, $D_k$ and compute an estimate $H_k$ of the F.I.M.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 1&lt;br /&gt;
|text=Consider the model&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_i {{!}} \psi_i &amp;amp;\sim&amp;amp; \pcyipsii(y_i {{!}} \psi_i) \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\Omega = {\rm diag}(\omega_1^2,\omega_2^2,\ldots,\omega_d^2)$ is a diagonal matrix and $h(\psi_i)=(h_1(\psi_{i,1}), h_2(\psi_{i,2}), \ldots , h_d(\psi_{i,d}) )^{\transpose}$.&lt;br /&gt;
The vector of population parameters is $\theta = (\psi_{\rm pop} , \Omega)=(\psi_{ {\rm pop},1},\ldots,\psi_{ {\rm pop},d},\omega_1^2,\ldots,\omega_d^2)$.&lt;br /&gt;
&lt;br /&gt;
Here,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log (\pyipsii(y_i,\psi_i;\theta)) = \log (\pcyipsii(y_i {{!}} \psi_i)) + \log (\ppsii(\psi_i;\theta)).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Dt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \Dt{\log (\ppsii(\psi_i;\theta))} \\&lt;br /&gt;
\DDt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \DDt{\log (\ppsii(\psi_i;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
More precisely,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\log (\ppsii(\psi_i;\theta)) &amp;amp;=&amp;amp; -\displaystyle{\frac{d}{2} }\log(2\pi) + \sum_{\iparam=1}^d \log(h_\iparam^{\prime}(\psi_{i,\iparam}))&lt;br /&gt;
-\displaystyle{ \frac{1}{2} } \sum_{\iparam=1}^d \log(\omega_\iparam^2)&lt;br /&gt;
-\sum_{\iparam=1}^d \displaystyle{ \frac{1}{2\, \omega_\iparam^2} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
\displaystyle{\frac{1}{\omega_\iparam^2} }h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) ) \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
-\displaystyle{ \frac{1}{2\omega_\iparam^2} }&lt;br /&gt;
+\displaystyle{\frac{1}{2\, \omega_\iparam^4} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \psi_{ {\rm pop},\jparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
 \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{\omega_\iparam^2} --&amp;gt;&lt;br /&gt;
\left( h_\iparam^{\prime\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )- h_\iparam^{\prime \, 2}(\psi_{ {\rm pop},\iparam}) \right)/\omega_\iparam^2 &amp;amp; {\rm if \quad } \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{2\omega_\iparam^4} - \frac{1}{\omega_\iparam^6} --&amp;gt;&lt;br /&gt;
1/(2\omega_\iparam^4) -&lt;br /&gt;
( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2/\omega_\iparam^6 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %      -\frac{1}{\omega_\iparam^4} --&amp;gt;&lt;br /&gt;
-h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )/\omega_\iparam^4 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 2&lt;br /&gt;
|text= We consider the same  model for continuous data, assuming a constant error model and  that the variance $a^2$ of the residual error has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $\theta_y=a^2$, $\theta_\psi=(\psi_{\rm pop},\Omega)$ and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pyipsii(y_i,\psi_i;\theta)) = \log(\pcyipsii(y_i {{!}} \psi_i ; a^2)) + \log(\ppsii(\psi_i;\psi_{\rm pop},\Omega)),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))$ with respect to $a^2$ are straightforward to compute. Derivatives of $\log(\ppsii(\psi_i;\psi_{\rm pop},\Omega))$ with respect to $\psi_{\rm pop}$ and $\Omega$ remain unchanged.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 3&lt;br /&gt;
|text=  Consider again the same  model for continuous data,  assuming now that a subset $\xi$ of the parameters of the structural model has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i,\xi) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\psi$ remain as the subset of individual parameters with variability. Here, $\theta_y=(\xi,a^2)$, $\theta_\psi=(\psi_{\rm pop},\Omega)$, and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; \xi,a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2 a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i,\xi))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; \xi, a^2))$ with respect to $\xi$ require  computation of the derivative of $f$ with respect to $\xi$. These derivatives are usually not calculable. One possibility is to numerically approximate them using finite differences.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation using linearization of the model == &lt;br /&gt;
&lt;br /&gt;
Consider here a model for continuous data that uses a $\phi$-parametrization for the individual parameters:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;= &amp;amp; f(t_{ij} , \phi_i) + g(t_{ij} , \phi_i)\teps_{ij} \\&lt;br /&gt;
\phi_i &amp;amp;=&amp;amp; \phi_{\rm pop} + \eta_i .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\hphi_i$ be some predicted value of $\phi_i$, such as for instance the estimated mean or  estimated mode of the conditional distribution $\pmacro(\phi_i |y_i ; \hat{\theta})$.&lt;br /&gt;
&lt;br /&gt;
We can then choose to linearize the model for the observations $(y_{ij}, 1\leq j \leq n_i)$ of individual $i$ around the vector of predicted individual parameters. Let $\Dphi{f(t , \phi)}$ be the row vector of derivatives of  $f(t , \phi)$ with respect to $\phi$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) + \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_i - \hphi_i)  + g(t_{ij} , \hphi_i)\teps_{ij} \\&lt;br /&gt;
&amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) +  \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i)&lt;br /&gt;
+  \Dphi{f(t_{ij} , \hphi_i)} \, \eta_i  + g(t_{ij} , \hphi_i)\teps_{ij} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, we can approximate the marginal distribution of the vector $y_i$ as a normal distribution:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_approx&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{i} \approx {\cal N}\left(f(t_{i} , \hphi_i) +  \Dphi{f(t_{i} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i) ,&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} \Omega \Dphi{f(t_{i} , \hphi_i)}^{\transpose}  + g(t_{i} , \hphi_i)\Sigma_{n_i} g(t_{ij} , \hphi_i)^{\transpose} \right),&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&lt;br /&gt;
where $\Sigma_{n_i}$ is the variance-covariance matrix of $\teps_{i,1},\ldots,\teps_{i,n_i}$. If the $\teps_{ij}$ are i.i.d., then&lt;br /&gt;
$\Sigma_{n_i}$ is the identity matrix.&lt;br /&gt;
&lt;br /&gt;
We can equivalently use the original $\psi$-parametrization and the fact that $\phi_i=h(\psi_i)$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} = \Dpsi{f(t_{i} , \hpsi_i)} J_h(\hpsi_i)^{\transpose} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $J_h$ is the Jacobian of $h$.&lt;br /&gt;
&lt;br /&gt;
We then can approximate the observed log-likelihood ${\llike}(\theta) = \log(\like(\theta;\by))=\sum_{i=1}^N \log(\pyi(y_i;\theta))$ using this normal approximation. We can also derive the F.I.M. by computing the matrix of second-order partial derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&lt;br /&gt;
Except for very simple models, computing these second-order partial derivatives in  closed form is not straightforward. In such cases, finite differences can be used for numerically approximating  them. We can use for instance a central difference approximation of the second derivative of $\llike(\theta)$. To this end, let $\nu&amp;gt;0$. For $j=1,2,\ldots, m$, let $\nu^{(j)}=(\nu^{(j)}_{k}, 1\leq k \leq m)$ be the $m$-vector such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\nu^{(j)}_{k} = \left\{&lt;br /&gt;
                  \begin{array}{ll}&lt;br /&gt;
                    \nu &amp;amp; {\rm if \quad j= k} \\&lt;br /&gt;
                    0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
                  \end{array}&lt;br /&gt;
                \right.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, for $\nu$ small enough,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial_{\theta_j}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{ \frac{ {\llike}(\theta+\nu^{(j)})- {\llike}(\theta-\nu^{(j)})}{2\nu} }  \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_diff&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial^2_{\theta_j,\theta_k}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{\frac{ {\llike}(\theta+\nu^{(j)}+\nu^{(k)})- {\llike}(\theta+\nu^{(j)}-\nu^{(k)})&lt;br /&gt;
-{\llike}(\theta-\nu^{(j)}+\nu^{(k)})+{\llike}(\theta-\nu^{(j)}-\nu^{(k)})}{4\nu^2} } . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, the algorithm for approximating the Fisher Information Matrix $I(\hat{\theta)}$ using a linear approximation of the model consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, obtain some estimate $(\hpsi_i)$ of the individual parameters $(\psi_i)$ (we can average for example the final terms of the sequence $(\psi_i^{(k)})$ drawn during the final iterations of the [[The SAEM algorithm for estimating population parameters| SAEM algorithm]]).&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. For $i=1,2,\ldots,N$, compute $\hphi_i=h(\hpsi_i)$,  the mean and the variance of the normal distribution defined in [[#eq:fim_approx|(5)]], and  ${\llike}(\theta)$ using this normal approximation.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3. Use  [[#eq:fim_diff|(6)]] to approximate the matrix of second-order derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkNext=Estimation of the log-likelihood&lt;br /&gt;
|linkBack=The Metropolis-Hastings algorithm for simulating the individual parameters }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7463</id>
		<title>Estimation of the observed Fisher information matrix</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Estimation_of_the_observed_Fisher_information_matrix&amp;diff=7463"/>
		<updated>2013-08-28T10:04:58Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Estimation using stochastic approximation */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;==Estimation using stochastic approximation==&lt;br /&gt;
&lt;br /&gt;
The ''observed'' Fisher information matrix (F.I.M.) is a function of $\theta$ defined as&lt;br /&gt;
  &lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq_fim1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
I(\theta) &amp;amp;=&amp;amp; -\DDt{\log ({\like}(\theta;\by))} \\&lt;br /&gt;
&amp;amp;=&amp;amp; -\DDt{\log (\py(\by;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
Due to the likelihood being quite complex, $I(\theta)$ usually has no closed form expression. It is however possible to estimate it using a stochastic approximation procedure based on &amp;lt;balloon title=&amp;quot;Kuhn05: put here the reference!!!&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;Louis' formula&amp;lt;/balloon&amp;gt;:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\DDt{\log (\pmacro(\by;\theta))} = \esp{\DDt{\log (\pmacro(\by,\bpsi;\theta))}| \by ;\theta} + \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta},&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
 \cov{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta} &amp;amp;=&amp;amp;&lt;br /&gt;
  \esp{ \left(\Dt{\log (\pmacro(\by,\bpsi;\theta))} \right)\left(\Dt{\log (\pmacro(\by,\bpsi;\theta))}\right)^{\transpose} | \by ; \theta} \\&lt;br /&gt;
&amp;amp;&amp;amp; - \esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta}\esp{\Dt{\log (\pmacro(\by,\bpsi;\theta))} | \by ; \theta}^{\transpose} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus, $\DDt{\log (\pmacro(\by;\theta))}$  is defined as a combination of conditional expectations. Each of these conditional expectations can be estimated by Monte Carlo, or equivalently approximated using a stochastic approximation algorithm.&lt;br /&gt;
&lt;br /&gt;
We can then draw a sequence  $(\psi_i^{(k)})$ using a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hasting algorithm]] and estimate the observed F.I.M. online. At iteration $k$ of the algorithm:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Simulation step''': for $i=1,2,\ldots,N$, draw $\psi_i^{(k)}$ from $m$ iterations of the Metropolis-Hastings algorithm described in [[The Metropolis-Hastings algorithm for simulating the individual parameters| The Metropolis-Hastings algorithm]] section with $\pmacro(\psi_i |y_i ;{\theta})$ as the limit distribution.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Stochastic approximation''': update $D_k$, $G_k$ and $\Delta_k$  according to the following recurrence relations:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k &amp;amp; = &amp;amp; \Delta_{k-1} + \gamma_k \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - \Delta_{k-1} \right) \\&lt;br /&gt;
D_k &amp;amp; = &amp;amp; D_{k-1} + \gamma_k \left(\DDt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))} - D_{k-1} \right)\\&lt;br /&gt;
G_k &amp;amp; = &amp;amp; G_{k-1} + \gamma_k \left((\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})(\Dt{\log (\pmacro(\by,\bpsi^{(k)};{\theta}))})^\transpose -G_{k-1} \right),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $(\gamma_k)$ is a decreasing sequence of positive numbers such that $\gamma_1=1$, $ \sum_{k=1}^{\infty} \gamma_k = \infty$,  and $\sum_{k=1}^{\infty} \gamma_k^2 &amp;lt; \infty$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* '''Estimation step''': update the estimate $H_k$ of the F.I.M. according to&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;H_k  =  D_k + G_k - \Delta_k \Delta_k^{\transpose}. &amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Implementing this algorithm therefore requires computation of the first and second derivatives of&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log (\pmacro(\by,\bpsi;\theta))=\sum_{i=1}^{N} \log (\pmacro(y_i,\psi_i;\theta)).&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Assume first that the joint distribution of $\by$ and $\bpsi$ decomposes as&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_dec1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pypsi(\by,\bpsi;\theta) = \pcypsi(\by {{!}} \bpsi)\ppsi(\bpsi;\theta).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
This assumption means that for any $i=1,2,\ldots,N$, all of the components of $\psi_i$ are random and  there exists a sufficient statistic ${\cal S}(\bpsi)$ for the estimation of $\theta$. It is then sufficient to compute the first and second derivatives of $\log (\pmacro(\bpsi;\theta))$ in order to estimate the F.I.M. This can be done relatively simply in closed form when the individual parameters are normally distributed (or a transformation $h$ of them is).&lt;br /&gt;
&lt;br /&gt;
If some component of $\psi_i$ has no variability, [[#eq:fim_dec1|(2)]] no longer holds, but we can decompose $\theta$ into $(\theta_y,\theta_\psi)$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) = \pcyipsii(y_i {{!}} \psi_i ; \theta_y)\ppsii(\psi_i;\theta_\psi).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We  then need to compute the first and second derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ and $\log(\ppsii(\psi_i;\theta_\psi))$. Derivatives of $\log(\pcyipsii(y_i |\psi_i ; \theta_y))$ that do not have a closed form expression can be obtained using central differences.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
1. Using $\gamma_k=1/k$ for $k \geq 1$ means that each term is approximated with an empirical mean obtained from $(\bpsi^{(k)}, k \geq 1)$. For instance,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Delta_k&lt;br /&gt;
&amp;amp;=&amp;amp;  \Delta_{k-1} + \displaystyle{ \frac{1}{k} } \left(\Dt{\log (\pmacro(\by,\bpsi^{(k)};\theta))} - \Delta_{k-1} \right)   &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_Delta2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
&amp;amp;=&amp;amp;  \displaystyle{ \frac{1}{k} }\sum_{j=1}^{k} \Dt{\log (\pmacro(\by,\bpsi^{(j)};\theta))} .  &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
[[#eq:fim_Delta1|(3)]] (resp. [[#eq:fim_Delta2|(4)]]) defines $\Delta_k$ using an online (resp. offline) algorithm. Writing $\Delta_k$ as in [[#eq:fim_Delta1|(3)]] instead of [[#eq:fim_Delta2|(4)]] avoids having to store all simulated sequences $(\bpsi^{(j)}, 1\leq j \leq k)$ when computing $\Delta_k$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2. This approach is used  for computing the F.I.M.  $I(\hat{\theta})$ in practice, where $\hat{\theta}$ is the maximum likelihood estimate of $\theta$. The only difference with the [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings]] used for SAEM is that the population parameter $\theta$ is not updated and remains fixed at  $\hat{\theta}$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, a stochastic approximation algorithm for estimating the observed Fisher Information Matrix $I(\hat{\theta)}$ consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, run a [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings algorithm]] to draw a sequence $\psi_i^{(k)}$ with limit distribution $\pmacro(\psi_i {{!}}y_i ;\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. At iteration $k$ of the Metropolis-Hastings algorithm, compute the first and second derivatives of $\pypsi(\by,\bpsi^{(k)};\hat{\theta})$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3.Update $\Delta_k$, $G_k$, $D_k$ and compute an estimate $H_k$ of the F.I.M.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 1&lt;br /&gt;
|text=Consider the model&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_i {{!}} \psi_i &amp;amp;\sim&amp;amp; \pcyipsii(y_i {{!}} \psi_i) \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega),&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\Omega = {\rm diag}(\omega_1^2,\omega_2^2,\ldots,\omega_d^2)$ is a diagonal matrix and $h(\psi_i)=(h_1(\psi_{i,1}), h_2(\psi_{i,2}), \ldots , h_d(\psi_{i,d}) )^{\transpose}$.&lt;br /&gt;
The vector of population parameters is $\theta = (\psi_{\rm pop} , \Omega)=(\psi_{ {\rm pop},1},\ldots,\psi_{ {\rm pop},d},\omega_1^2,\ldots,\omega_d^2)$.&lt;br /&gt;
&lt;br /&gt;
Here,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log (\pyipsii(y_i,\psi_i;\theta)) = \log (\pcyipsii(y_i {{!}} \psi_i)) + \log (\ppsii(\psi_i;\theta)).&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\Dt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \Dt{\log (\ppsii(\psi_i;\theta))} \\&lt;br /&gt;
\DDt{\log (\pyipsii(y_i,\psi_i;\theta))} &amp;amp;=&amp;amp; \DDt{\log (\ppsii(\psi_i;\theta))} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
More precisely,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\log (\ppsii(\psi_i;\theta)) &amp;amp;=&amp;amp; -\displaystyle{\frac{d}{2} }\log(2\pi) + \sum_{\iparam=1}^d \log(h_\iparam^{\prime}(\psi_{i,\iparam}))&lt;br /&gt;
-\displaystyle{ \frac{1}{2} } \sum_{\iparam=1}^d \log(\omega_\iparam^2)&lt;br /&gt;
-\sum_{\iparam=1}^d \displaystyle{ \frac{1}{2\, \omega_\iparam^2} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
\displaystyle{\frac{1}{\omega_\iparam^2} }h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) ) \\&lt;br /&gt;
\partial \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
-\displaystyle{ \frac{1}{2\omega_\iparam^2} }&lt;br /&gt;
+\displaystyle{\frac{1}{2\, \omega_\iparam^4} }( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \psi_{ {\rm pop},\jparam}  &amp;amp;=&amp;amp;&lt;br /&gt;
 \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{\omega_\iparam^2} --&amp;gt;&lt;br /&gt;
\left( h_\iparam^{\prime\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )- h_\iparam^{\prime \, 2}(\psi_{ {\rm pop},\iparam}) \right)/\omega_\iparam^2 &amp;amp; {\rm if \quad } \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
 \\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \omega^2_{\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %     \frac{1}{2\omega_\iparam^4} - \frac{1}{\omega_\iparam^6} --&amp;gt;&lt;br /&gt;
1/(2\omega_\iparam^4) -&lt;br /&gt;
( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )^2/\omega_\iparam^6 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\\&lt;br /&gt;
\partial^2 \log (\ppsii(\psi_i;\theta))/\partial \psi_{ {\rm pop},\iparam} \partial \omega^2_{\jparam} &amp;amp;=&amp;amp; \left\{&lt;br /&gt;
   \begin{array}{ll}&lt;br /&gt;
&amp;lt;!-- %      -\frac{1}{\omega_\iparam^4} --&amp;gt;&lt;br /&gt;
-h_\iparam^{\prime}(\psi_{ {\rm pop},\iparam})( h_\iparam(\psi_{i,\iparam}) - h_\iparam(\psi_{ {\rm pop},\iparam}) )/\omega_\iparam^4 &amp;amp; {\rm if \quad} \iparam=\jparam \\&lt;br /&gt;
     0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
   \end{array}&lt;br /&gt;
 \right.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 2&lt;br /&gt;
|text= We consider the same  model for continuous data, assuming a constant error model and  that the variance $a^2$ of the residual error has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $\theta_y=a^2$, $\theta_\psi=(\psi_{\rm pop},\Omega)$ and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pyipsii(y_i,\psi_i;\theta)) = \log(\pcyipsii(y_i {{!}} \psi_i ; a^2)) + \log(\ppsii(\psi_i;\psi_{\rm pop},\Omega)),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; a^2))$ with respect to $a^2$ are straightforward to compute. Derivatives of $\log(\ppsii(\psi_i;\psi_{\rm pop},\Omega))$ with respect to $\psi_{\rm pop}$ and $\Omega$ remain unchanged.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 3&lt;br /&gt;
|text=  Consider again the same  model for continuous data,  assuming now that a subset $\xi$ of the parameters of the structural model has no variability:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} {{!}} \psi_i &amp;amp;\sim&amp;amp; {\cal N}(f(t_{ij}, \psi_i,\xi) \ , \ a^2), \ \ 1 \leq j \leq n_i \\&lt;br /&gt;
h(\psi_i) &amp;amp;\sim_{i.i.d}&amp;amp; {\cal N}( h(\psi_{\rm pop}) , \Omega).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\psi$ remain as the subset of individual parameters with variability. Here, $\theta_y=(\xi,a^2)$, $\theta_\psi=(\psi_{\rm pop},\Omega)$, and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\log(\pcyipsii(y_i {{!}} \psi_i ; \xi,a^2))&lt;br /&gt;
=-\displaystyle{\frac{n_i}{2} }\log(2\pi)- \displaystyle{\frac{n_i}{2} }\log(a^2) - \displaystyle{\frac{1}{2 a^2} }\sum_{j=1}^{n_i}(y_{ij} - f(t_{ij}, \psi_i,\xi))^2 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Derivatives of $\log(\pcyipsii(y_i {{!}} \psi_i ; \xi, a^2))$ with respect to $\xi$ require  computation of the derivative of $f$ with respect to $\xi$. These derivatives are usually not calculable. One possibility is to numerically approximate them using finite differences.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation using linearization of the model == &lt;br /&gt;
&lt;br /&gt;
Consider here a model for continuous data that uses a $\phi$-parametrization for the individual parameters:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;= &amp;amp; f(t_{ij} , \phi_i) + g(t_{ij} , \phi_i)\teps_{ij} \\&lt;br /&gt;
\phi_i &amp;amp;=&amp;amp; \phi_{\rm pop} + \eta_i .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Let $\hphi_i$ be some predicted value of $\phi_i$, such as for instance the estimated mean or  estimated mode of the conditional distribution $\pmacro(\phi_i |y_i ; \hat{\theta})$.&lt;br /&gt;
&lt;br /&gt;
We can then choose to linearize the model for the observations $(y_{ij}, 1\leq j \leq n_i)$ of individual $i$ around the vector of predicted individual parameters. Let $\Dphi{f(t , \phi)}$ be the row vector of derivatives of  $f(t , \phi)$ with respect to $\phi$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) + \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_i - \hphi_i)  + g(t_{ij} , \hphi_i)\teps_{ij} \\&lt;br /&gt;
&amp;amp;\simeq&amp;amp; f(t_{ij} , \hphi_i) +  \Dphi{f(t_{ij} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i)&lt;br /&gt;
+  \Dphi{f(t_{ij} , \hphi_i)} \, \eta_i  + g(t_{ij} , \hphi_i)\teps_{ij} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, we can approximate the marginal distribution of the vector $y_i$ as a normal distribution:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_approx&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
y_{i} \approx {\cal N}\left(f(t_{i} , \hphi_i) +  \Dphi{f(t_{i} , \hphi_i)} \, (\phi_{\rm pop} - \hphi_i) ,&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} \Omega \Dphi{f(t_{i} , \hphi_i)}^{\transpose}  + g(t_{i} , \hphi_i)\Sigma_{n_i} g(t_{ij} , \hphi_i)^{\transpose} \right),&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&lt;br /&gt;
where $\Sigma_{n_i}$ is the variance-covariance matrix of $\teps_{i,1},\ldots,\teps_{i,n_i}$. If the $\teps_{ij}$ are i.i.d., then&lt;br /&gt;
$\Sigma_{n_i}$ is the identity matrix.&lt;br /&gt;
&lt;br /&gt;
We can equivalently use the original $\psi$-parametrization and the fact that $\phi_i=h(\psi_i)$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\Dphi{f(t_{i} , \hphi_i)} = \Dpsi{f(t_{i} , \hpsi_i)} J_h(\hpsi_i)^{\transpose} , &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $J_h$ is the Jacobian of $h$.&lt;br /&gt;
&lt;br /&gt;
We then can approximate the observed log-likelihood ${\llike}(\theta) = \log(\like(\theta;\by))=\sum_{i=1}^N \log(\pyi(y_i;\theta))$ using this normal approximation. We can also derive the F.I.M. by computing the matrix of second-order partial derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&lt;br /&gt;
Except for very simple models, computing these second-order partial derivatives in  closed form is not straightforward. In such cases, finite differences can be used for numerically approximating  them. We can use for instance a central difference approximation of the second derivative of $\llike(\theta)$. To this end, let $\nu&amp;gt;0$. For $j=1,2,\ldots, m$, let $\nu^{(j)}=(\nu^{(j)}_{k}, 1\leq k \leq m)$ be the $m$-vector such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\nu^{(j)}_{k} = \left\{&lt;br /&gt;
                  \begin{array}{ll}&lt;br /&gt;
                    \nu &amp;amp; {\rm if \quad j= k} \\&lt;br /&gt;
                    0 &amp;amp; {\rm otherwise.}&lt;br /&gt;
                  \end{array}&lt;br /&gt;
                \right.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Then, for $\nu$ small enough,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial_{\theta_j}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{ \frac{ {\llike}(\theta+\nu^{(j)})- {\llike}(\theta-\nu^{(j)})}{2\nu} }  \\&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:fim_diff&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\partial^2_{\theta_j,\theta_k}{ {\llike}(\theta)} &amp;amp;\approx&amp;amp; \displaystyle{\frac{ {\llike}(\theta+\nu^{(j)}+\nu^{(k)})- {\llike}(\theta+\nu^{(j)}-\nu^{(k)})&lt;br /&gt;
-{\llike}(\theta-\nu^{(j)}+\nu^{(k)})+{\llike}(\theta-\nu^{(j)}-\nu^{(k)})}{4\nu^2} } . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=In summary, for a given estimate $\hat{\theta}$ of the population parameter $\theta$, the algorithm for approximating the Fisher Information Matrix $I(\hat{\theta)}$ using a linear approximation of the model consists of:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
1. For $i=1,2,\ldots,N$, obtain some estimate $(\hpsi_i)$ of the individual parameters $(\psi_i)$ (we can average for example the final terms of the sequence $(\psi_i^{(k)})$ drawn during the final iterations of the [[The SAEM algorithm for estimating population parameters| SAEM algorithm]]).&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
2. For $i=1,2,\ldots,N$, compute $\hphi_i=h(\hpsi_i)$,  the mean and the variance of the normal distribution defined in [[#eq:fim_approx|(5)]], and  ${\llike}(\theta)$ using this normal approximation.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
3. Use  [[#eq:fim_diff|(6)]] to approximate the matrix of second-order derivatives of ${\llike}(\theta)$.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkNext=Estimation of the log-likelihood&lt;br /&gt;
|linkBack=The Metropolis-Hastings algorithm for simulating the individual parameters }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7462</id>
		<title>Model for categorical data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7462"/>
		<updated>2013-08-28T09:42:11Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
== Overview == &lt;br /&gt;
&lt;br /&gt;
Assume now that the observed data  takes its values in a fixed and finite set of nominal categories $\{c_1, c_2,\ldots , c_K\}$.&lt;br /&gt;
Considering the observations $(y_{ij}, 1 \leq j \leq n_i)$ of any individual $i$ as a sequence of independent random variables, the model is completely defined by the probability mass functions $\prob{y_{ij}=c_k | \psi_i}$, for $k=1,\ldots, K$ and $1 \leq j \leq n_i$.&lt;br /&gt;
&lt;br /&gt;
For a given $(i,j)$, the sum of the $K$ probabilities is 1, so in fact only $K-1$ of them need to be defined.&lt;br /&gt;
&lt;br /&gt;
In the most general way possible, any model can be considered so long as it defines a probability distribution, i.e., for each $k$,  $\prob{y_{ij}=c_k | \psi_i} \in [0,1]$, and $\sum_{k=1}^{K} \prob{y_{ij}=c_k | \psi_i} = 1$. For instance, we could define $K$ time-dependent parametric functions $a_1$, $a_2$, ..., $a_K$  and set for any individual $i$, time $t_{ij}$ and $k \in \{1,\ldots,K\}$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;categorical1&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\prob{y_{ij}=c_k {{!}} \psi_i} = \displaystyle{\frac{e^{a_k(t_{ij},\psi_i)} }{\sum_{m=1}^K e^{a_m(t_{ij},\psi_i)} } }.   &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= Suppose we want to model binary data, i.e., data where $y_{ij} \in \{0,1\}$.&lt;br /&gt;
&lt;br /&gt;
Let $\psi_i=(\alpha_i,\beta_i)$ and let $a_1(t,\psi_i)=0$ and  $a_2(t,\psi_i) = \alpha_i + \beta_i \, t$. Then, [[#categorical1|(1)]] gives a probability distribution for binary outcomes:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation= &amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij}=0 {{!}} \psi_i} = \displaystyle{\frac{1}{1 + e^{\alpha_i + \beta_i \, t_{ij} } } } \quad \ \ \ \text{and} \quad&lt;br /&gt;
\ \ \ \prob{y_{ij}=1 {{!}} \psi_i} = \displaystyle{\frac{e^{\alpha_i + \beta_i \, t_{ij} } }{1 + e^{\alpha_i + \beta_i \, t_{ij} } } }. &lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Such parametrizations are extremely flexible and easy to interpret in simple situations.&lt;br /&gt;
In the previous example for instance, $\prob{y_{ij}=1 | \psi_i}$ and $a_2(t_{ij},\psi_i)$ move in the same direction as time increases.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Ordinal data ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Ordinal data further assumes that the categories are ordered, i.e., there exists an order $\prec$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
c_1 \prec c_2,\prec \ldots \prec c_K .&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
We can think for instance of levels of pain (low, moderate, severe), or any scores on a discrete scale, e.g., from 1 to 10.&lt;br /&gt;
&lt;br /&gt;
Instead of defining the probabilities of each category, it may be convenient to define the cumulative probabilities $\prob{y_{ij} \preceq c_k | \psi_i}$ for $k=1,\ldots ,K-1$, or in the other direction: $\prob{y_{ij} \succeq c_k | \psi_i}$ for $k=2,\ldots, K$. &lt;br /&gt;
Any model is possible as long as it defines a probability distribution, i.e., satisfies:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
0 \leq  \prob{y_{ij} \preceq c_1 {{!}} \psi_i} \leq  \prob{y_{ij} \preceq c_2 {{!}} \bpsi_i} \leq \ldots \leq  \prob{y_{ij} \preceq c_K {{!}} \psi_i} =1 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Without any loss of generality, we will consider numerical categories in what follows. The order $\prec$  then reduces to the usual order $&amp;lt;$ on $\Rset$.&lt;br /&gt;
&lt;br /&gt;
Currently, the most popular model for  ordinal data is the proportional odds model which uses ''logits'' of these cumulative probabilities, also called ''cumulative logits''. We assume that there exist $\alpha_{i,1}\geq0$, $\alpha_{i,2}\geq 0, \ldots , \alpha_{i,K-1}\geq 0$ such that for $k=1,2,\ldots,K-1$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq c_k {{!}} \psi_i} \right) = \left( \sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij}) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
where $x(t_{ij})$ is a vector of regression variables and $\beta_i$ a vector of coefficients. Here, $\bpsi_i=(\alpha_{i1},\alpha_{i2},\ldots,\alpha_{i,K-1},\beta_i)$.&lt;br /&gt;
&lt;br /&gt;
Recall that $\logit(p) = \log\left(p/(1-p)\right)$. Then, the probability defined in [[#propodds_model|(2)]] can also be expressed as&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij} \leq c_k {{!}} \bpsi_i}  = \displaystyle{\frac{1}{1 + e^{ \left(\sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij})} } }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= We give  to patients a drug which is supposed to decrease the level of a given type of pain. &lt;br /&gt;
The level of pain is measured on a scale from 1 to 3: 1=low, 2=moderate, 3=high. We consider the following model with the constraint that $\alpha_{i2}\geq 0$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 1 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \beta_{i,1}\, t_{ij} + \beta_{i,2}\, C_{ij} \\&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 2 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \alpha_{i,2} + \beta_{i,1}\, t_{ij} + \beta_{i2}\, C_{ij} \\&lt;br /&gt;
\prob{y_{ij} \leq 3 {{!}} \psi_i} &amp;amp;=&amp;amp; 1,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
where $C_{ij}$ is the concentration of the drug at time $t_{ij}$. The model parameters are quite easy to explain:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $\beta_{i,1}=0$   means that without treatment, the level of pain tends to remains stable  over time.&lt;br /&gt;
* $\beta_{i,1}&amp;lt;0$  (resp. $\beta_{i1}&amp;gt;0$) means that the pain tends to increase (resp. decrease) over time.&lt;br /&gt;
* $\beta_{i,2}=0$   means that the drug has no effect on pain.&lt;br /&gt;
* $\beta_{i,2}&amp;gt;0$   means that the level of pain tends to decrease when the  drug concentration increases, whereas $\beta_{i2}&amp;lt;0$ means that pain is an adverse drug effect.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Exclusive use of linear models (or generalized linear models) has no real justification today since very efficient tools are available for nonlinear models.&lt;br /&gt;
Model [[#propodds_model|(2)]] can be easily extended to a nonlinear model:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model2&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq k {{!}} \psi_i } \right) = \sum_{m=1}^k \alpha_{i,m} +  \beta(x(t_{ij})) , &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $\beta$ is any (linear or nonlinear) function of $x(t_{ij})$. }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Markovian dependence ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity, we will assume here that the observations $(y_{ij})$ take their values in $\{1, 2, \ldots, K\}$.&lt;br /&gt;
&lt;br /&gt;
We have so far assumed that the categorical observations $(y_{ij},\,j=1,2,\ldots,n_i)$ for individual $i$ are independent. It is however possible to introduce dependency between observations from the same individual by assuming that $(y_{ij},\,j=1,2,\ldots,n_i)$ forms a [http://en.wikipedia.org/wiki/Markov_chain Markov chain]. For instance, a Markov chain with memory 1 assumes that all is required from the past  to determine the distribution of $y_{i,j}$ is the value of  the previous observation $y_{i,j-1}$. i.e., for all $k=1,2,\ldots ,K$,&lt;br /&gt;
 &lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i,j} = k\, {{!}} \,y_{i,j-1}, y_{i,j-2}, y_{i,j-3},\ldots,\psi_i} = \prob{y_{i,j} = k {{!}} y_{i,j-1},\psi_i}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Discrete time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
If the observation times are regularly spaced (constant length of time between successive observations), we can consider the observations $(y_{ij},\,j=1,2,\ldots,n_i)$ to be a discrete time Markov chain. Here, for each individual $i$, the probability distribution of the sequence $(y_{ij},\,j=1,2,\ldots,n_i)$ is defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the distribution $  \pi_{i,1} = (\pi_{i,1}^{k} , k=1,2,\ldots,K)$ of the first observation $y_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \pi_{i,1}^{k} = \prob{y_{i,1} = k {{!}} \psi_i} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the sequence of ''transition matrices'' $(Q_{i,j}, j=2,3,\ldots)$, where for each $j$, $Q_{i,j} = (q_{i,j}^{\ell,k}, 1\leq \ell,k \leq K)$ is a matrix of size $K \times K$ such that,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
q_{i,j}^{\ell,k} &amp;amp;=&amp;amp; \prob{y_{i,j} = k {{!}} y_{i,j-1}=\ell , \psi_i} \quad \text{ for all } (\ell,k),\\&lt;br /&gt;
\sum_{k=1}^{K}q_{ij}^{\ell,k} &amp;amp;=&amp;amp; 1 \quad \text{ for all } (\ell,k).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution of  $y_i=(y_{i,j}, j=1,2,\ldots, n_i)$ is then well-defined:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \pmacro(y_{i,1}{{!}}\psi_i) \prod_{j=2}^{n_i} \pmacro(y_{i,j} {{!}} y_{i,j-1},\psi_i) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
For a given individual $i$,  $Q_{i,j}$ defines the transition probabilities between states at a given time $t_{ij}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:markov_1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Our model must therefore give, for each individual $i$, the distribution of first observation $(y_{i,1})$ and a description of how the transition probabilities evolve with time.&lt;br /&gt;
&lt;br /&gt;
The figure below shows several examples of simulated sequences coming from a model with 2 states defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit\left(q_{i,j}^{1,2}\right) &amp;amp;=&amp;amp; a_i+b_i \, t_j \\&lt;br /&gt;
\logit\left(q_{i,j}^{2,1}\right) &amp;amp;=&amp;amp; c_i+d_i \, t_j \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5 ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $t_j = j$.&lt;br /&gt;
&lt;br /&gt;
[[File:markov_2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
In the first example (left), the logits of the transitions between states are constant ($b_i = d_i = 0$).&lt;br /&gt;
Transition probabilities are therefore constant over time.  Here,  $q^{1,2}=1/(1+\exp(2.5))=0.0759$ and $q^{2,1}=1/(1+\exp(2))=0.1192$. As $q^{1,2}$ and $q^{2,1}$ are small with $q^{1,2}&amp;lt;q^{2,1}$, transitions between the two states are rare, and a larger amount of time (on average) is spent in  state 1. Indeed, the stationary distribution is the eigenvector of the transition matrix $P$: $\prob{y_{ij}=1}=0.611$ and $ \prob{y_{ij}=2}=0.389$.&lt;br /&gt;
The figure (left) displays the transition rates $q^{1,2}$ and $q^{2,1}$ as function of the time (top left) and two simulated sequences of states (centre and bottom left).&lt;br /&gt;
&lt;br /&gt;
In the second example (center), $b_i$ and $d_i$ are negative. This means that as time progresses, transitions from state 1 to 2 become rarer, and the same is true from 2 to 1.&lt;br /&gt;
&lt;br /&gt;
In the third example (right), now $b_i$ and $d_i$ are positive. This means that as time progresses, transitions from state 1 to 2 become more and more frequent, and also more frequent from 2 to 1.&lt;br /&gt;
&lt;br /&gt;
Note that the value of $a_i$ (resp. $c_i$) can be seen as the transition probability from state 1 to 2 (resp. 2 to 1) at time $t=0$.&lt;br /&gt;
&lt;br /&gt;
Different choices can be made for defining an initial distribution $\pi_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The initial state can be defined arbitrarily: $y_{i,1}=k_0$. This means that $\pi_{i,1}^{k_0} = 1$ and $\pi_{i,1}^{k} = 0$ for $k\neq k_0$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* More generally, any simple probability distribution can be put on the choice of the initial state, e.g., the uniform distribution $\pi_{i,1}^{k} = 1/K$ for $ k=1,2,\ldots , K$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If a transition matrix $Q_{i1} $ has been defined at time $t_1$, we might consider using its stationary distribution, i.e., taking for $\pi_{i,1}$ the solution to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pi_{i,1} = \pi_{i,1} Q_{i1} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Continuous time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The previous situation can be extended to the case where observation times are irregular, by modeling the&lt;br /&gt;
sequence of states as a continuous-time [http://en.wikipedia.org/wiki/Markov_process Markov process]. The difference is that rather than transitioning to a new (possibly the same) state at each time step, the system remains in the current state for some random  amount of time before transitioning. This process is now  characterized by ''transition rates'' instead of transition probabilities:&lt;br /&gt;
&lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(t+h) = k\, {{!}} \,y_{i}(t)=\ell , \psi_i} = h \, \rho_{i}^{\ell,k}(t) + o(h),\quad k \neq \ell .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The probability that no transition happens between $t$ and $t+h$ is&lt;br /&gt;
 &lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(s) = \ell, \forall  s\in(t, t+h) \ {{!}}  \ y_{i}(t)=\ell , \psi_i} = e^{h \, \rho_{i}^{\ell,\ell}(t)} . &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Furthermore, for any individual $i$ and any time $t$, the transition rates $\rho_{i}^{\ell,k}(t)$ satisfy&lt;br /&gt;
&lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
 \sum_{k=1}^K \rho_{i}^{\ell,k}(t) = 0 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------------------------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary&lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=  &lt;br /&gt;
A model for independent categorical data is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;The probability mass functions $\left(\prob{y_{ij} = k {{!}} \psi_i} \right)$&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative probability functions $\left(\prob{y_{ij} \leq c_k {{!}}  \psi_i} \right)$ for ordinal data&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative logits $\left(\logit \left( \prob{y_{ij} \leq k {{!}}  \psi_i} \right)\right)$ for a proportional odds model&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
A model for categorical data with Markovian dependency is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the probability transitions in the case of a discrete-time Markov chain&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the transition rates in the case of a continuous-time Markov process&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; the probability distribution of the initial states&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== $\mlxtran$ for categorical data models == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 1:&lt;br /&gt;
|title2= $ \quad y_{ij} \in \{0, 1, 2\}$&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (V_i, k_i, \alpha_{0,i}, \alpha_{1,i}, \gamma_i) \\[0.2cm]&lt;br /&gt;
D &amp;amp;=&amp;amp;100 \\&lt;br /&gt;
C(t,\psi_i) &amp;amp;=&amp;amp; \frac{D_i}{V_i} e^{-k_i \, t} \\[0.2cm]&lt;br /&gt;
\prob{y_{ij}\leq 0} &amp;amp;=&amp;amp; \alpha_{0,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 1} &amp;amp;=&amp;amp; \alpha_{0,i} + \alpha_{1,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 2} &amp;amp;=&amp;amp; 1&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;&lt;br /&gt;
INPUT:&lt;br /&gt;
input = {V, k, alpha0, alpha1, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
D = 100&lt;br /&gt;
C = D/V*exp(-k*t)&lt;br /&gt;
p0 = alpha0 + gamma*C&lt;br /&gt;
p1 = p0 + alpha1&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
y = {type=categorical,&lt;br /&gt;
     categories={0, 1, 2},&lt;br /&gt;
     P(y&amp;lt;=0)=p0,&lt;br /&gt;
     P(y&amp;lt;=1)=p1&lt;br /&gt;
     }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 2:&lt;br /&gt;
|title2= $\quad$ 2-state discrete-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i) \\[0.2cm]&lt;br /&gt;
\logit(p_{ij}^{12}) &amp;amp;=&amp;amp; a_i+b_i \, t_{ij} \\&lt;br /&gt;
\logit(p_{ij}^{21}) &amp;amp;=&amp;amp; c_i+d_i \, t_{ij} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = 0.5&lt;br /&gt;
      logit(P(Y=2 | Y_p=1)) = a + b*t&lt;br /&gt;
      logit(P(Y=1 | Y_p=2)) = c + d*t&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 3:&lt;br /&gt;
|title2= $\quad$ 2-state continuous-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i,\pi_i) \\[0.2cm]&lt;br /&gt;
q_{i}^{12}(t) &amp;amp;=&amp;amp; e^{a_i+b_i \, t} \\&lt;br /&gt;
q_{i}^{21}(t) &amp;amp;=&amp;amp; e^{c_i+d_i \, t} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; \pi_i&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d, pi}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = pi&lt;br /&gt;
      transitionRate(1,2) = exp(a + b*t)&lt;br /&gt;
      transitionRate(2,1) = exp(c + d*t)&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2010analysis,&lt;br /&gt;
  title={Analysis of ordinal categorical data},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={656},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2007introduction,&lt;br /&gt;
  title={An introduction to categorical data analysis},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={423},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bolker2009generalized,&lt;br /&gt;
  title={Generalized linear mixed models: a practical guide for ecology and evolution},&lt;br /&gt;
  author={Bolker, B. M. and Brooks, M. E. and Clark, C. J. and Geange, S. W. and Poulsen, J. R. and Stevens, M. H. H. and White, J.-S. S. and others},&lt;br /&gt;
  journal={Trends in ecology &amp;amp; evolution},&lt;br /&gt;
  volume={24},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={127-135},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Elsevier Science}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{davidian1995,&lt;br /&gt;
        author = {Davidian, M. and Giltinan, D. M.},&lt;br /&gt;
        title = {Nonlinear Models for Repeated Measurements Data },&lt;br /&gt;
        publisher = {Chapman &amp;amp; Hall.},&lt;br /&gt;
        address = {London},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        year = {1995}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{jiang2007,&lt;br /&gt;
        author = {Jiang., J.},&lt;br /&gt;
        title  = {Linear and Generalized Linear Mixed Models and Their Applications.},&lt;br /&gt;
        publisher = {Springer Series in Statistics},&lt;br /&gt;
        volume = {},&lt;br /&gt;
        pages = {},&lt;br /&gt;
        year = {2007},&lt;br /&gt;
        series = {},&lt;br /&gt;
        address = {New York},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        month = {}&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mcculloch2011generalized,&lt;br /&gt;
  title={Generalized, Linear, and Mixed Models},&lt;br /&gt;
  author={McCulloch, C. E. and Searle, S. R. and Neuhaus, J. M.},&lt;br /&gt;
  isbn={9781118209967},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
  url={http://books.google.fr/books?id=kyvgyK\_sBlkC},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{molenberghs2005models,&lt;br /&gt;
  title={Models for discrete longitudinal data},&lt;br /&gt;
  author={Molenberghs, G. and Verbeke, G.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{powers2008statistical,&lt;br /&gt;
  title={Statistical methods for categorical data analysis},&lt;br /&gt;
  author={Powers, D. A. and Xie, Y.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Emerald Group Publishing}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wolfinger1993generalized,&lt;br /&gt;
  title={Generalized linear mixed models a pseudo-likelihood approach},&lt;br /&gt;
  author={Wolfinger, R. and O'Connell, M.},&lt;br /&gt;
  journal={Journal of statistical Computation and Simulation},&lt;br /&gt;
  volume={48},&lt;br /&gt;
  number={3-4},&lt;br /&gt;
  pages={233-243},&lt;br /&gt;
  year={1993},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Models for count data&lt;br /&gt;
|linkNext=Models for time-to-event data }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7461</id>
		<title>Model for categorical data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7461"/>
		<updated>2013-08-28T09:39:25Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
== Overview == &lt;br /&gt;
&lt;br /&gt;
Assume now that the observed data  takes its values in a fixed and finite set of nominal categories $\{c_1, c_2,\ldots , c_K\}$.&lt;br /&gt;
Considering the observations $(y_{ij}, 1 \leq j \leq n_i)$ of any individual $i$ as a sequence of independent random variables, the model is completely defined by the probability mass functions $\prob{y_{ij}=c_k | \psi_i}$, for $k=1,\ldots, K$ and $1 \leq j \leq n_i$.&lt;br /&gt;
&lt;br /&gt;
For a given $(i,j)$, the sum of the $K$ probabilities is 1, so in fact only $K-1$ of them need to be defined.&lt;br /&gt;
&lt;br /&gt;
In the most general way possible, any model can be considered so long as it defines a probability distribution, i.e., for each $k$,  $\prob{y_{ij}=c_k | \psi_i} \in [0,1]$, and $\sum_{k=1}^{K} \prob{y_{ij}=c_k | \psi_i} = 1$. For instance, we could define $K$ time-dependent parametric functions $a_1$, $a_2$, ..., $a_K$  and set for any individual $i$, time $t_{ij}$ and $k \in \{1,\ldots,K\}$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;categorical1&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\prob{y_{ij}=c_k {{!}} \psi_i} = \displaystyle{\frac{e^{a_k(t_{ij},\psi_i)} }{\sum_{m=1}^K e^{a_m(t_{ij},\psi_i)} } }.   &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= Suppose we want to model binary data, i.e., data where $y_{ij} \in \{0,1\}$.&lt;br /&gt;
&lt;br /&gt;
Let $\psi_i=(\alpha_i,\beta_i)$ and let $a_1(t,\psi_i)=0$ and  $a_2(t,\psi_i) = \alpha_i + \beta_i \, t$. Then, [[#categorical1|(1)]] gives a probability distribution for binary outcomes:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation= &amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij}=0 {{!}} \psi_i} = \displaystyle{\frac{1}{1 + e^{\alpha_i + \beta_i \, t_{ij} } } } \quad \ \ \ \text{and} \quad&lt;br /&gt;
\ \ \ \prob{y_{ij}=1 {{!}} \psi_i} = \displaystyle{\frac{e^{\alpha_i + \beta_i \, t_{ij} } }{1 + e^{\alpha_i + \beta_i \, t_{ij} } } }. &lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Such parametrizations are extremely flexible and easy to interpret in simple situations.&lt;br /&gt;
In the previous example for instance, $\prob{y_{ij}=1 | \psi_i}$ and $a_2(t_{ij},\psi_i)$ move in the same direction as time increases.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Ordinal data ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Ordinal data further assumes that the categories are ordered, i.e., there exists an order $\prec$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
c_1 \prec c_2,\prec \ldots \prec c_K .&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
We can think for instance of levels of pain (low, moderate, severe), or any scores on a discrete scale, e.g., from 1 to 10.&lt;br /&gt;
&lt;br /&gt;
Instead of defining the probabilities of each category, it may be convenient to define the cumulative probabilities $\prob{y_{ij} \preceq c_k | \psi_i}$ for $k=1,\ldots ,K-1$, or in the other direction: $\prob{y_{ij} \succeq c_k | \psi_i}$ for $k=2,\ldots, K$. &lt;br /&gt;
Any model is possible as long as it defines a probability distribution, i.e., satisfies:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
0 \leq  \prob{y_{ij} \preceq c_1 {{!}} \psi_i} \leq  \prob{y_{ij} \preceq c_2 {{!}} \bpsi_i} \leq \ldots \leq  \prob{y_{ij} \preceq c_K {{!}} \psi_i} =1 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Without any loss of generality, we will consider numerical categories in what follows. The order $\prec$  then reduces to the usual order $&amp;lt;$ on $\Rset$.&lt;br /&gt;
&lt;br /&gt;
Currently, the most popular model for  ordinal data is the proportional odds model which uses ''logits'' of these cumulative probabilities, also called ''cumulative logits''. We assume that there exist $\alpha_{i,1}\geq0$, $\alpha_{i,2}\geq 0, \ldots , \alpha_{i,K-1}\geq 0$ such that for $k=1,2,\ldots,K-1$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq c_k {{!}} \psi_i} \right) = \left( \sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij}) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
where $x(t_{ij})$ is a vector of regression variables and $\beta_i$ a vector of coefficients. Here, $\bpsi_i=(\alpha_{i1},\alpha_{i2},\ldots,\alpha_{i,K-1},\beta_i)$.&lt;br /&gt;
&lt;br /&gt;
Recall that $\logit(p) = \log\left(p/(1-p)\right)$. Then, the probability defined in [[#propodds_model|(2)]] can also be expressed as&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij} \leq c_k {{!}} \bpsi_i}  = \displaystyle{\frac{1}{1 + e^{ \left(\sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij})} } }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= We give  to patients a drug which is supposed to decrease the level of a given type of pain. &lt;br /&gt;
The level of pain is measured on a scale from 1 to 3: 1=low, 2=moderate, 3=high. We consider the following model with the constraint that $\alpha_{i2}\geq 0$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 1 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \beta_{i,1}\, t_{ij} + \beta_{i,2}\, C_{ij} \\&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 2 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \alpha_{i,2} + \beta_{i,1}\, t_{ij} + \beta_{i2}\, C_{ij} \\&lt;br /&gt;
\prob{y_{ij} \leq 3 {{!}} \psi_i} &amp;amp;=&amp;amp; 1,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
where $C_{ij}$ is the concentration of the drug at time $t_{ij}$. The model parameters are quite easy to explain:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $\beta_{i,1}=0$   means that without treatment, the level of pain tends to remains stable  over time.&lt;br /&gt;
* $\beta_{i,1}&amp;lt;0$  (resp. $\beta_{i1}&amp;gt;0$) means that the pain tends to increase (resp. decrease) over time.&lt;br /&gt;
* $\beta_{i,2}=0$   means that the drug has no effect on pain.&lt;br /&gt;
* $\beta_{i,2}&amp;gt;0$   means that the level of pain tends to decrease when the  drug concentration increases, whereas $\beta_{i2}&amp;lt;0$ means that pain is an adverse drug effect.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Exclusive use of linear models (or generalized linear models) has no real justification today since very efficient tools are available for nonlinear models.&lt;br /&gt;
Model [[#propodds_model|(2)]] can be easily extended to a nonlinear model:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model2&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq k {{!}} \psi_i } \right) = \sum_{m=1}^k \alpha_{i,m} +  \beta(x(t_{ij})) , &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $\beta$ is any (linear or nonlinear) function of $x(t_{ij})$. }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Markovian dependence ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity, we will assume here that the observations $(y_{ij})$ take their values in $\{1, 2, \ldots, K\}$.&lt;br /&gt;
&lt;br /&gt;
We have so far assumed that the categorical observations $(y_{ij},\,j=1,2,\ldots,n_i)$ for individual $i$ are independent. It is however possible to introduce dependency between observations from the same individual by assuming that $(y_{ij},\,j=1,2,\ldots,n_i)$ forms a [http://en.wikipedia.org/wiki/Markov_chain Markov chain]. For instance, a Markov chain with memory 1 assumes that all is required from the past  to determine the distribution of $y_{i,j}$ is the value of  the previous observation $y_{i,j-1}$. i.e., for all $k=1,2,\ldots ,K$,&lt;br /&gt;
 &lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i,j} = k\, {{!}} \,y_{i,j-1}, y_{i,j-2}, y_{i,j-3},\ldots,\psi_i} = \prob{y_{i,j} = k {{!}} y_{i,j-1},\psi_i}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Discrete time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
If the observation times are regularly spaced (constant length of time between successive observations), we can consider the observations $(y_{ij},\,j=1,2,\ldots,n_i)$ to be a discrete time Markov chain. Here, for each individual $i$, the probability distribution of the sequence $(y_{ij},\,j=1,2,\ldots,n_i)$ is defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the distribution $  \pi_{i,1} = (\pi_{i,1}^{k} , k=1,2,\ldots,K)$ of the first observation $y_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \pi_{i,1}^{k} = \prob{y_{i,1} = k {{!}} \psi_i} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the sequence of ''transition matrices'' $(Q_{i,j}, j=2,3,\ldots)$, where for each $j$, $Q_{i,j} = (q_{i,j}^{\ell,k}, 1\leq \ell,k \leq K)$ is a matrix of size $K \times K$ such that,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
q_{i,j}^{\ell,k} &amp;amp;=&amp;amp; \prob{y_{i,j} = k {{!}} y_{i,j-1}=\ell , \psi_i} \quad \text{ for all } (\ell,k),\\&lt;br /&gt;
\sum_{k=1}^{K}q_{ij}^{\ell,k} &amp;amp;=&amp;amp; 1 \quad \text{ for all } (\ell,k).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution of  $y_i=(y_{i,j}, j=1,2,\ldots, n_i)$ is then well-defined:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \pmacro(y_{i,1}{{!}}\psi_i) \prod_{j=2}^{n_i} \pmacro(y_{i,j} {{!}} y_{i,j-1},\psi_i) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
For a given individual $i$,  $Q_{i,j}$ defines the transition probabilities between states at a given time $t_{ij}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:markov_1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Our model must therefore give, for each individual $i$, the distribution of first observation $(y_{i,1})$ and a description of how the transition probabilities evolve with time.&lt;br /&gt;
&lt;br /&gt;
The figure below shows several examples of simulated sequences coming from a model with 2 states defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit\left(q_{i,j}^{1,2}\right) &amp;amp;=&amp;amp; a_i+b_i \, t_j \\&lt;br /&gt;
\logit\left(q_{i,j}^{2,1}\right) &amp;amp;=&amp;amp; c_i+d_i \, t_j \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5 ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $t_j = j$.&lt;br /&gt;
&lt;br /&gt;
[[File:markov_2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
In the first example (left), the logits of the transitions between states are constant ($b_i = d_i = 0$).&lt;br /&gt;
Transition probabilities are therefore constant over time.  Here,  $q^{1,2}=1/(1+\exp(2.5))=0.0759$ and $q^{2,1}=1/(1+\exp(2))=0.1192$. As $q^{1,2}$ and $q^{2,1}$ are small with $q^{1,2}&amp;lt;q^{2,1}$, transitions between the two states are rare, and a larger amount of time (on average) is spent in  state 1. Indeed, the stationary distribution is the eigenvector of the transition matrix $P$: $\prob{y_{ij}=1}=0.611$ and $ \prob{y_{ij}=2}=0.389$.&lt;br /&gt;
The figure (left) displays the transition rates $q^{1,2}$ and $q^{2,1}$ as function of the time (top left) and two simulated sequences of states (centre and bottom left).&lt;br /&gt;
&lt;br /&gt;
In the second example (center), $b_i$ and $d_i$ are negative. This means that as time progresses, transitions from state 1 to 2 become rarer, and the same is true from 2 to 1.&lt;br /&gt;
&lt;br /&gt;
In the third example (right), now $b_i$ and $d_i$ are positive. This means that as time progresses, transitions from state 1 to 2 become more and more frequent, and also more frequent from 2 to 1.&lt;br /&gt;
&lt;br /&gt;
Note that the value of $a_i$ (resp. $c_i$) can be seen as the transition probability from state 1 to 2 (resp. 2 to 1) at time $t=0$.&lt;br /&gt;
&lt;br /&gt;
Different choices can be made for defining an initial distribution $\pi_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The initial state can be defined arbitrarily: $y_{i,1}=k_0$. This means that $\pi_{i,1}^{k_0} = 1$ and $\pi_{i,1}^{k} = 0$ for $k\neq k_0$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* More generally, any simple probability distribution can be put on the choice of the initial state, e.g., the uniform distribution $\pi_{i,1}^{k} = 1/K$ for $ k=1,2,\ldots , K$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If a transition matrix $Q_{i1} $ has been defined at time $t_1$, we might consider using its stationary distribution, i.e., taking for $\pi_{i,1}$ the solution to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pi_{i,1} = \pi_{i,1} Q_{i1} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Continuous time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The previous situation can be extended to the case where observation times are irregular, by modeling the&lt;br /&gt;
sequence of states as a continuous-time [http://en.wikipedia.org/wiki/Markov_process Markov process]. The difference is that rather than transitioning to a new (possibly the same) state at each time step, the system remains in the current state for some random  amount of time before transitioning. This process is now  characterized by ''transition rates'' instead of transition probabilities:&lt;br /&gt;
&lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(t+h) = k\, {{!}} \,y_{i}(t)=\ell , \psi_i} = h \, \rho_{i}^{\ell,k}(t) + o(h),\quad k \neq \ell .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The probability that no transition happens between $t$ and $t+h$ is&lt;br /&gt;
 &lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(s) = \ell, \forall  s\in(t, t+h) \ {{!}}  \ y_{i}(t)=\ell , \psi_i} = e^{h \, \rho_{i}^{\ell,\ell}(t)} . &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------------------------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary&lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=  &lt;br /&gt;
A model for independent categorical data is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;The probability mass functions $\left(\prob{y_{ij} = k {{!}} \psi_i} \right)$&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative probability functions $\left(\prob{y_{ij} \leq c_k {{!}}  \psi_i} \right)$ for ordinal data&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative logits $\left(\logit \left( \prob{y_{ij} \leq k {{!}}  \psi_i} \right)\right)$ for a proportional odds model&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
A model for categorical data with Markovian dependency is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the probability transitions in the case of a discrete-time Markov chain&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the transition rates in the case of a continuous-time Markov process&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; the probability distribution of the initial states&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== $\mlxtran$ for categorical data models == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 1:&lt;br /&gt;
|title2= $ \quad y_{ij} \in \{0, 1, 2\}$&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (V_i, k_i, \alpha_{0,i}, \alpha_{1,i}, \gamma_i) \\[0.2cm]&lt;br /&gt;
D &amp;amp;=&amp;amp;100 \\&lt;br /&gt;
C(t,\psi_i) &amp;amp;=&amp;amp; \frac{D_i}{V_i} e^{-k_i \, t} \\[0.2cm]&lt;br /&gt;
\prob{y_{ij}\leq 0} &amp;amp;=&amp;amp; \alpha_{0,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 1} &amp;amp;=&amp;amp; \alpha_{0,i} + \alpha_{1,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 2} &amp;amp;=&amp;amp; 1&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;&lt;br /&gt;
INPUT:&lt;br /&gt;
input = {V, k, alpha0, alpha1, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
D = 100&lt;br /&gt;
C = D/V*exp(-k*t)&lt;br /&gt;
p0 = alpha0 + gamma*C&lt;br /&gt;
p1 = p0 + alpha1&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
y = {type=categorical,&lt;br /&gt;
     categories={0, 1, 2},&lt;br /&gt;
     P(y&amp;lt;=0)=p0,&lt;br /&gt;
     P(y&amp;lt;=1)=p1&lt;br /&gt;
     }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 2:&lt;br /&gt;
|title2= $\quad$ 2-state discrete-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i) \\[0.2cm]&lt;br /&gt;
\logit(p_{ij}^{12}) &amp;amp;=&amp;amp; a_i+b_i \, t_{ij} \\&lt;br /&gt;
\logit(p_{ij}^{21}) &amp;amp;=&amp;amp; c_i+d_i \, t_{ij} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = 0.5&lt;br /&gt;
      logit(P(Y=2 | Y_p=1)) = a + b*t&lt;br /&gt;
      logit(P(Y=1 | Y_p=2)) = c + d*t&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 3:&lt;br /&gt;
|title2= $\quad$ 2-state continuous-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i,\pi_i) \\[0.2cm]&lt;br /&gt;
q_{i}^{12}(t) &amp;amp;=&amp;amp; e^{a_i+b_i \, t} \\&lt;br /&gt;
q_{i}^{21}(t) &amp;amp;=&amp;amp; e^{c_i+d_i \, t} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; \pi_i&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d, pi}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = pi&lt;br /&gt;
      transitionRate(1,2) = exp(a + b*t)&lt;br /&gt;
      transitionRate(2,1) = exp(c + d*t)&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2010analysis,&lt;br /&gt;
  title={Analysis of ordinal categorical data},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={656},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2007introduction,&lt;br /&gt;
  title={An introduction to categorical data analysis},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={423},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bolker2009generalized,&lt;br /&gt;
  title={Generalized linear mixed models: a practical guide for ecology and evolution},&lt;br /&gt;
  author={Bolker, B. M. and Brooks, M. E. and Clark, C. J. and Geange, S. W. and Poulsen, J. R. and Stevens, M. H. H. and White, J.-S. S. and others},&lt;br /&gt;
  journal={Trends in ecology &amp;amp; evolution},&lt;br /&gt;
  volume={24},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={127-135},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Elsevier Science}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{davidian1995,&lt;br /&gt;
        author = {Davidian, M. and Giltinan, D. M.},&lt;br /&gt;
        title = {Nonlinear Models for Repeated Measurements Data },&lt;br /&gt;
        publisher = {Chapman &amp;amp; Hall.},&lt;br /&gt;
        address = {London},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        year = {1995}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{jiang2007,&lt;br /&gt;
        author = {Jiang., J.},&lt;br /&gt;
        title  = {Linear and Generalized Linear Mixed Models and Their Applications.},&lt;br /&gt;
        publisher = {Springer Series in Statistics},&lt;br /&gt;
        volume = {},&lt;br /&gt;
        pages = {},&lt;br /&gt;
        year = {2007},&lt;br /&gt;
        series = {},&lt;br /&gt;
        address = {New York},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        month = {}&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mcculloch2011generalized,&lt;br /&gt;
  title={Generalized, Linear, and Mixed Models},&lt;br /&gt;
  author={McCulloch, C. E. and Searle, S. R. and Neuhaus, J. M.},&lt;br /&gt;
  isbn={9781118209967},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
  url={http://books.google.fr/books?id=kyvgyK\_sBlkC},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{molenberghs2005models,&lt;br /&gt;
  title={Models for discrete longitudinal data},&lt;br /&gt;
  author={Molenberghs, G. and Verbeke, G.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{powers2008statistical,&lt;br /&gt;
  title={Statistical methods for categorical data analysis},&lt;br /&gt;
  author={Powers, D. A. and Xie, Y.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Emerald Group Publishing}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wolfinger1993generalized,&lt;br /&gt;
  title={Generalized linear mixed models a pseudo-likelihood approach},&lt;br /&gt;
  author={Wolfinger, R. and O'Connell, M.},&lt;br /&gt;
  journal={Journal of statistical Computation and Simulation},&lt;br /&gt;
  volume={48},&lt;br /&gt;
  number={3-4},&lt;br /&gt;
  pages={233-243},&lt;br /&gt;
  year={1993},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Models for count data&lt;br /&gt;
|linkNext=Models for time-to-event data }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7460</id>
		<title>Model for categorical data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7460"/>
		<updated>2013-08-28T09:31:15Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
== Overview == &lt;br /&gt;
&lt;br /&gt;
Assume now that the observed data  takes its values in a fixed and finite set of nominal categories $\{c_1, c_2,\ldots , c_K\}$.&lt;br /&gt;
Considering the observations $(y_{ij}, 1 \leq j \leq n_i)$ of any individual $i$ as a sequence of independent random variables, the model is completely defined by the probability mass functions $\prob{y_{ij}=c_k | \psi_i}$, for $k=1,\ldots, K$ and $1 \leq j \leq n_i$.&lt;br /&gt;
&lt;br /&gt;
For a given $(i,j)$, the sum of the $K$ probabilities is 1, so in fact only $K-1$ of them need to be defined.&lt;br /&gt;
&lt;br /&gt;
In the most general way possible, any model can be considered so long as it defines a probability distribution, i.e., for each $k$,  $\prob{y_{ij}=c_k | \psi_i} \in [0,1]$, and $\sum_{k=1}^{K} \prob{y_{ij}=c_k | \psi_i} = 1$. For instance, we could define $K$ time-dependent parametric functions $a_1$, $a_2$, ..., $a_K$  and set for any individual $i$, time $t_{ij}$ and $k \in \{1,\ldots,K\}$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;categorical1&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\prob{y_{ij}=c_k {{!}} \psi_i} = \displaystyle{\frac{e^{a_k(t_{ij},\psi_i)} }{\sum_{m=1}^K e^{a_m(t_{ij},\psi_i)} } }.   &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= Suppose we want to model binary data, i.e., data where $y_{ij} \in \{0,1\}$.&lt;br /&gt;
&lt;br /&gt;
Let $\psi_i=(\alpha_i,\beta_i)$ and let $a_1(t,\psi_i)=0$ and  $a_2(t,\psi_i) = \alpha_i + \beta_i \, t$. Then, [[#categorical1|(1)]] gives a probability distribution for binary outcomes:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation= &amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij}=0 {{!}} \psi_i} = \displaystyle{\frac{1}{1 + e^{\alpha_i + \beta_i \, t_{ij} } } } \quad \ \ \ \text{and} \quad&lt;br /&gt;
\ \ \ \prob{y_{ij}=1 {{!}} \psi_i} = \displaystyle{\frac{e^{\alpha_i + \beta_i \, t_{ij} } }{1 + e^{\alpha_i + \beta_i \, t_{ij} } } }. &lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Such parametrizations are extremely flexible and easy to interpret in simple situations.&lt;br /&gt;
In the previous example for instance, $\prob{y_{ij}=1 | \psi_i}$ and $a_2(t_{ij},\psi_i)$ move in the same direction as time increases.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Ordinal data ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Ordinal data further assumes that the categories are ordered, i.e., there exists an order $\prec$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
c_1 \prec c_2,\prec \ldots \prec c_K .&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
We can think for instance of levels of pain (low, moderate, severe), or any scores on a discrete scale, e.g., from 1 to 10.&lt;br /&gt;
&lt;br /&gt;
Instead of defining the probabilities of each category, it may be convenient to define the cumulative probabilities $\prob{y_{ij} \preceq c_k | \psi_i}$ for $k=1,\ldots ,K-1$, or in the other direction: $\prob{y_{ij} \succeq c_k | \psi_i}$ for $k=2,\ldots, K$. &lt;br /&gt;
Any model is possible as long as it defines a probability distribution, i.e., satisfies:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
0 \leq  \prob{y_{ij} \preceq c_1 {{!}} \psi_i} \leq  \prob{y_{ij} \preceq c_2 {{!}} \bpsi_i} \leq \ldots \leq  \prob{y_{ij} \preceq c_K {{!}} \psi_i} =1 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Without any loss of generality, we will consider numerical categories in what follows. The order $\prec$  then reduces to the usual order $&amp;lt;$ on $\Rset$.&lt;br /&gt;
&lt;br /&gt;
Currently, the most popular model for  ordinal data is the proportional odds model which uses ''logits'' of these cumulative probabilities, also called ''cumulative logits''. We assume that there exist $\alpha_{i,1}\geq0$, $\alpha_{i,2}\geq 0, \ldots , \alpha_{i,K-1}\geq 0$ such that for $k=1,2,\ldots,K-1$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq c_k {{!}} \psi_i} \right) = \left( \sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij}) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
where $x(t_{ij})$ is a vector of regression variables and $\beta_i$ a vector of coefficients. Here, $\bpsi_i=(\alpha_{i1},\alpha_{i2},\ldots,\alpha_{i,K-1},\beta_i)$.&lt;br /&gt;
&lt;br /&gt;
Recall that $\logit(p) = \log\left(p/(1-p)\right)$. Then, the probability defined in [[#propodds_model|(2)]] can also be expressed as&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij} \leq c_k {{!}} \bpsi_i}  = \displaystyle{\frac{1}{1 + e^{ \left(\sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij})} } }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= We give  to patients a drug which is supposed to decrease the level of a given type of pain. &lt;br /&gt;
The level of pain is measured on a scale from 1 to 3: 1=low, 2=moderate, 3=high. We consider the following model with the constraint that $\alpha_{i2}\geq 0$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 1 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \beta_{i,1}\, t_{ij} + \beta_{i,2}\, C_{ij} \\&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 2 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \alpha_{i,2} + \beta_{i,1}\, t_{ij} + \beta_{i2}\, C_{ij} \\&lt;br /&gt;
\prob{y_{ij} \leq 3 {{!}} \psi_i} &amp;amp;=&amp;amp; 1,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
where $C_{ij}$ is the concentration of the drug at time $t_{ij}$. The model parameters are quite easy to explain:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $\beta_{i,1}=0$   means that without treatment, the level of pain tends to remains stable  over time.&lt;br /&gt;
* $\beta_{i,1}&amp;lt;0$  (resp. $\beta_{i1}&amp;gt;0$) means that the pain tends to increase (resp. decrease) over time.&lt;br /&gt;
* $\beta_{i,2}=0$   means that the drug has no effect on pain.&lt;br /&gt;
* $\beta_{i,2}&amp;gt;0$   means that the level of pain tends to decrease when the  drug concentration increases, whereas $\beta_{i2}&amp;lt;0$ means that pain is an adverse drug effect.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Exclusive use of linear models (or generalized linear models) has no real justification today since very efficient tools are available for nonlinear models.&lt;br /&gt;
Model [[#propodds_model|(2)]] can be easily extended to a nonlinear model:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model2&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq k {{!}} \psi_i } \right) = \sum_{m=1}^k \alpha_{i,m} +  \beta(x(t_{ij})) , &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $\beta$ is any (linear or nonlinear) function of $x(t_{ij})$. }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Markovian dependence ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity, we will assume here that the observations $(y_{ij})$ take their values in $\{1, 2, \ldots, K\}$.&lt;br /&gt;
&lt;br /&gt;
We have so far assumed that the categorical observations $(y_{ij},\,j=1,2,\ldots,n_i)$ for individual $i$ are independent. It is however possible to introduce dependency between observations from the same individual by assuming that $(y_{ij},\,j=1,2,\ldots,n_i)$ forms a [http://en.wikipedia.org/wiki/Markov_chain Markov chain]. For instance, a Markov chain with memory 1 assumes that all is required from the past  to determine the distribution of $y_{i,j}$ is the value of  the previous observation $y_{i,j-1}$. i.e., for all $k=1,2,\ldots ,K$,&lt;br /&gt;
 &lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i,j} = k\, {{!}} \,y_{i,j-1}, y_{i,j-2}, y_{i,j-3},\ldots,\psi_i} = \prob{y_{i,j} = k {{!}} y_{i,j-1},\psi_i}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Discrete time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
If the observation times are regularly spaced (constant length of time between successive observations), we can consider the observations $(y_{ij},\,j=1,2,\ldots,n_i)$ to be a discrete time Markov chain. Here, for each individual $i$, the probability distribution of the sequence $(y_{ij},\,j=1,2,\ldots,n_i)$ is defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the distribution $  \pi_{i,1} = (\pi_{i,1}^{k} , k=1,2,\ldots,K)$ of the first observation $y_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \pi_{i,1}^{k} = \prob{y_{i,1} = k {{!}} \psi_i} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the sequence of ''transition matrices'' $(Q_{i,j}, j=2,3,\ldots)$, where for each $j$, $Q_{i,j} = (q_{i,j}^{\ell,k}, 1\leq \ell,k \leq K)$ is a matrix of size $K \times K$ such that,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
q_{i,j}^{\ell,k} &amp;amp;=&amp;amp; \prob{y_{i,j} = k {{!}} y_{i,j-1}=\ell , \psi_i} \quad \text{ for all } (\ell,k),\\&lt;br /&gt;
\sum_{k=1}^{K}q_{ij}^{\ell,k} &amp;amp;=&amp;amp; 1 \quad \text{ for all } (\ell,k).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution of  $y_i=(y_{i,j}, j=1,2,\ldots, n_i)$ is then well-defined:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \pmacro(y_{i,1}{{!}}\psi_i) \prod_{j=2}^{n_i} \pmacro(y_{i,j} {{!}} y_{i,j-1},\psi_i) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
For a given individual $i$,  $Q_{i,j}$ defines the transition probabilities between states at a given time $t_{ij}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:markov_1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Our model must therefore give, for each individual $i$, the distribution of first observation $(y_{i,1})$ and a description of how the transition probabilities evolve with time.&lt;br /&gt;
&lt;br /&gt;
The figure below shows several examples of simulated sequences coming from a model with 2 states defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit\left(q_{i,j}^{1,2}\right) &amp;amp;=&amp;amp; a_i+b_i \, t_j \\&lt;br /&gt;
\logit\left(q_{i,j}^{2,1}\right) &amp;amp;=&amp;amp; c_i+d_i \, t_j \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5 ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $t_j = j$.&lt;br /&gt;
&lt;br /&gt;
[[File:markov_2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
In the first example (left), the logits of the transitions between states are constant ($b_i = d_i = 0$).&lt;br /&gt;
Transition probabilities are therefore constant over time.  Here,  $q^{1,2}=1/(1+\exp(2.5))=0.0759$ and $q^{2,1}=1/(1+\exp(2))=0.1192$. As $q^{1,2}$ and $q^{2,1}$ are small with $q^{1,2}&amp;lt;q^{2,1}$, transitions between the two states are rare, and a larger amount of time (on average) is spent in  state 1. Indeed, the stationary distribution is the eigenvector of the transition matrix $P$: $\prob{y_{ij}=1}=0.611$ and $ \prob{y_{ij}=2}=0.389$.&lt;br /&gt;
The figure (left) displays the transition rates $q^{1,2}$ and $q^{2,1}$ as function of the time (top left) and two simulated sequences of states (centre and bottom left).&lt;br /&gt;
&lt;br /&gt;
In the second example (center), $b_i$ and $d_i$ are negative. This means that as time progresses, transitions from state 1 to 2 become rarer, and the same is true from 2 to 1.&lt;br /&gt;
&lt;br /&gt;
In the third example (right), now $b_i$ and $d_i$ are positive. This means that as time progresses, transitions from state 1 to 2 become more and more frequent, and also more frequent from 2 to 1.&lt;br /&gt;
Note that the value of $a_i$ (resp. $c_i$) can be seen as the transition probability from state 1 to 2 (resp. 2 to 1) at time $t=0$.&lt;br /&gt;
&lt;br /&gt;
Different choices can be made for defining an initial distribution $\pi_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The initial state can be defined arbitrarily: $y_{i,1}=k_0$. This means that $\pi_{i,1}^{k_0} = 1$ and $\pi_{i,1}^{k} = 0$ for $k\neq k_0$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* More generally, any simple probability distribution can be put on the choice of the initial state, e.g., the uniform distribution $\pi_{i,1}^{k} = 1/K$ for $ k=1,2,\ldots , K$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If a transition matrix $Q_{i1} $ has been defined at time $t_1$, we might consider using its stationary distribution, i.e., taking for $\pi_{i,1}$ the solution to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pi_{i,1} = \pi_{i,1} Q_{i1} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Continuous time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The previous situation can be extended to the case where observation times are irregular, by modeling the&lt;br /&gt;
sequence of states as a continuous-time [http://en.wikipedia.org/wiki/Markov_process Markov process]. The difference is that rather than transitioning to a new (possibly the same) state at each time step, the system remains in the current state for some random  amount of time before transitioning. This process is now  characterized by ''transition rates'' instead of transition probabilities:&lt;br /&gt;
&lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(t+h) = k\, {{!}} \,y_{i}(t)=\ell , \psi_i} = h \, \rho_{i}^{\ell,k}(t) + o(h),\quad k \neq \ell .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The probability that no transition happens between $t$ and $t+h$ is&lt;br /&gt;
 &lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(s) = \ell, \forall  s\in(t, t+h) \ {{!}}  \ y_{i}(t)=\ell , \psi_i} = e^{h \, \rho_{i}^{\ell,\ell}(t)} . &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------------------------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary&lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=  &lt;br /&gt;
A model for independent categorical data is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;The probability mass functions $\left(\prob{y_{ij} = k {{!}} \psi_i} \right)$&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative probability functions $\left(\prob{y_{ij} \leq c_k {{!}}  \psi_i} \right)$ for ordinal data&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative logits $\left(\logit \left( \prob{y_{ij} \leq k {{!}}  \psi_i} \right)\right)$ for a proportional odds model&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
A model for categorical data with Markovian dependency is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the probability transitions in the case of a discrete-time Markov chain&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the transition rates in the case of a continuous-time Markov process&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; the probability distribution of the initial states&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== $\mlxtran$ for categorical data models == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 1:&lt;br /&gt;
|title2= $ \quad y_{ij} \in \{0, 1, 2\}$&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (V_i, k_i, \alpha_{0,i}, \alpha_{1,i}, \gamma_i) \\[0.2cm]&lt;br /&gt;
D &amp;amp;=&amp;amp;100 \\&lt;br /&gt;
C(t,\psi_i) &amp;amp;=&amp;amp; \frac{D_i}{V_i} e^{-k_i \, t} \\[0.2cm]&lt;br /&gt;
\prob{y_{ij}\leq 0} &amp;amp;=&amp;amp; \alpha_{0,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 1} &amp;amp;=&amp;amp; \alpha_{0,i} + \alpha_{1,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 2} &amp;amp;=&amp;amp; 1&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;&lt;br /&gt;
INPUT:&lt;br /&gt;
input = {V, k, alpha0, alpha1, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
D = 100&lt;br /&gt;
C = D/V*exp(-k*t)&lt;br /&gt;
p0 = alpha0 + gamma*C&lt;br /&gt;
p1 = p0 + alpha1&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
y = {type=categorical,&lt;br /&gt;
     categories={0, 1, 2},&lt;br /&gt;
     P(y&amp;lt;=0)=p0,&lt;br /&gt;
     P(y&amp;lt;=1)=p1&lt;br /&gt;
     }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 2:&lt;br /&gt;
|title2= $\quad$ 2-state discrete-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i) \\[0.2cm]&lt;br /&gt;
\logit(p_{ij}^{12}) &amp;amp;=&amp;amp; a_i+b_i \, t_{ij} \\&lt;br /&gt;
\logit(p_{ij}^{21}) &amp;amp;=&amp;amp; c_i+d_i \, t_{ij} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = 0.5&lt;br /&gt;
      logit(P(Y=2 | Y_p=1)) = a + b*t&lt;br /&gt;
      logit(P(Y=1 | Y_p=2)) = c + d*t&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 3:&lt;br /&gt;
|title2= $\quad$ 2-state continuous-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i,\pi_i) \\[0.2cm]&lt;br /&gt;
q_{i}^{12}(t) &amp;amp;=&amp;amp; e^{a_i+b_i \, t} \\&lt;br /&gt;
q_{i}^{21}(t) &amp;amp;=&amp;amp; e^{c_i+d_i \, t} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; \pi_i&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d, pi}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = pi&lt;br /&gt;
      transitionRate(1,2) = exp(a + b*t)&lt;br /&gt;
      transitionRate(2,1) = exp(c + d*t)&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2010analysis,&lt;br /&gt;
  title={Analysis of ordinal categorical data},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={656},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2007introduction,&lt;br /&gt;
  title={An introduction to categorical data analysis},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={423},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bolker2009generalized,&lt;br /&gt;
  title={Generalized linear mixed models: a practical guide for ecology and evolution},&lt;br /&gt;
  author={Bolker, B. M. and Brooks, M. E. and Clark, C. J. and Geange, S. W. and Poulsen, J. R. and Stevens, M. H. H. and White, J.-S. S. and others},&lt;br /&gt;
  journal={Trends in ecology &amp;amp; evolution},&lt;br /&gt;
  volume={24},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={127-135},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Elsevier Science}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{davidian1995,&lt;br /&gt;
        author = {Davidian, M. and Giltinan, D. M.},&lt;br /&gt;
        title = {Nonlinear Models for Repeated Measurements Data },&lt;br /&gt;
        publisher = {Chapman &amp;amp; Hall.},&lt;br /&gt;
        address = {London},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        year = {1995}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{jiang2007,&lt;br /&gt;
        author = {Jiang., J.},&lt;br /&gt;
        title  = {Linear and Generalized Linear Mixed Models and Their Applications.},&lt;br /&gt;
        publisher = {Springer Series in Statistics},&lt;br /&gt;
        volume = {},&lt;br /&gt;
        pages = {},&lt;br /&gt;
        year = {2007},&lt;br /&gt;
        series = {},&lt;br /&gt;
        address = {New York},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        month = {}&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mcculloch2011generalized,&lt;br /&gt;
  title={Generalized, Linear, and Mixed Models},&lt;br /&gt;
  author={McCulloch, C. E. and Searle, S. R. and Neuhaus, J. M.},&lt;br /&gt;
  isbn={9781118209967},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
  url={http://books.google.fr/books?id=kyvgyK\_sBlkC},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{molenberghs2005models,&lt;br /&gt;
  title={Models for discrete longitudinal data},&lt;br /&gt;
  author={Molenberghs, G. and Verbeke, G.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{powers2008statistical,&lt;br /&gt;
  title={Statistical methods for categorical data analysis},&lt;br /&gt;
  author={Powers, D. A. and Xie, Y.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Emerald Group Publishing}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wolfinger1993generalized,&lt;br /&gt;
  title={Generalized linear mixed models a pseudo-likelihood approach},&lt;br /&gt;
  author={Wolfinger, R. and O'Connell, M.},&lt;br /&gt;
  journal={Journal of statistical Computation and Simulation},&lt;br /&gt;
  volume={48},&lt;br /&gt;
  number={3-4},&lt;br /&gt;
  pages={233-243},&lt;br /&gt;
  year={1993},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Models for count data&lt;br /&gt;
|linkNext=Models for time-to-event data }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Model_evaluation&amp;diff=7454</id>
		<title>Model evaluation</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Model_evaluation&amp;diff=7454"/>
		<updated>2013-06-25T14:22:59Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Defining the expression &amp;quot;model evaluation&amp;quot; is a harder task than it may appear at first glance. Intuitively, it would seem to suggest evaluating the performance of a model based on the observed data, the same data that was used to build the model. Fair enough, but what then do we mean by the &amp;quot;performance&amp;quot; of a model?&lt;br /&gt;
&lt;br /&gt;
Do we mean the ability of the model to characterize and explain the phenomena being studied, in which case the goal is to use the model to understand the phenomena? Or do we mean the model's predictive performance when the model is used to predict the phenomena's behavior, either in the future or under new experimental conditions?&lt;br /&gt;
&lt;br /&gt;
What is comes down to is this: do we want to use the model to understand or to predict? This is the key question to ask before even starting to think about what tools to use and tasks to execute.&lt;br /&gt;
&lt;br /&gt;
Here, we will be for the most part focused on the ability of a model to explain the phenomena and data. Therefore, the first goal will be to check whether the data are in agreement with the model, and vice versa. In this process, model diagnostics can be used to eliminate model candidates that do not seem capable of reproducing the observed data.&lt;br /&gt;
As is the usual case in statistics, it is not because a model has not been rejected that it is necessarily the &amp;quot;true&amp;quot; one. All that we can say is that the experimental data does not allow us to reject this model. It is merely one of perhaps many models that cannot be rejected. Indeed, we can usually find several models that get past this first diagnostic step and are therefore not rejected.&lt;br /&gt;
&lt;br /&gt;
What to do, then, when several possible models are retained? Well, we can try to select the &amp;quot;best&amp;quot; one (or best ones if no leader distinguishes itself from the rest). This means developing a model selection process which allows us to compare the models to each other. Compare models? But with what criteria?&lt;br /&gt;
&lt;br /&gt;
In a purely explanatory context, [http://en.wikipedia.org/wiki/Occam%27s_razor Occam's razor] is a useful parsimony principle which states that among competing hypotheses, the one with the fewest assumptions should be selected. In a modeling context, this means that among valid competing  models, the most simple one should be selected.&lt;br /&gt;
&lt;br /&gt;
Model diagnostic tools are for the most part graphical or visual: we  &amp;quot;see&amp;quot; when something is not right between a chosen model and the data it is hypothesized to describe. Model selection tools, on the other hand, are analytical: we calculate the value of some criteria that allows us to compare the models with each other. However, it is absolutely critical to keep in mind the limits of these tools. These are not decision-making tools. It is not a $p$-value or some information criteria that can automatically decide which model to choose. It is always the modeler who must have the last word! This person uses the model diagnostic and selection tools in order to guide their decision, but at the end, it is they that must make the final decision. There is nothing more dangerous that rules applied without thinking and arbitrary cut-off values used blindly without reflection.&lt;br /&gt;
&lt;br /&gt;
Here, we will not look model selection techniques based on the various models' predictive performances. One such approach consists of splitting the data into three sets: a ''learning set'' is used for fitting the model, a ''validation set''  for choosing between models and a ''test set'' to assess the quality of the predictions made by the chosen model.&lt;br /&gt;
&lt;br /&gt;
Very few model diagnostic and selection tools exist for mixed-effects models. One of the most complete is [http://xpose.sourceforge.net Xpose], an R-based model-building aid for population analysis using [http://www.iconplc.com/technology/products/nonmem/ NONMEM]   that facilitates model diagnostics, candidate covariate identification and model comparison. Here we will use $\monolix$ and illustrate these techniques with the  example used previously for model exploration and parameter estimation.&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model diagnostics==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Model diagnostics  and statistical tests ===&lt;br /&gt;
&lt;br /&gt;
Suppose first that a model has been entirely defined by the modeler, and that its parameters have either been chosen or estimated.&lt;br /&gt;
What we call the &amp;quot;model&amp;quot; is therefore  a joint probability distribution along with some parameter values.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Note ${\cal M}_0$ the model we wish to evaluate. We place ourselves in the framework of statistical testing, and would like to perform the following hypothesis test:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; H_0: \quad {\cal M}={\cal M}_0 \quad vs \quad H_1: \quad {\cal M}\neq{\cal M}_0. &amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&amp;quot;Passing&amp;quot; the test does not mean that we accept $H_0$ but rather that we do not reject it. We will use the same point of view for model diagnostics whereby we eliminate model candidates that do not seem capable of reproducing the observed data, i.e., models for which we conclude that ${\cal M}\neq{\cal M}_0$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=Running a statistical test only makes sense when we have doubts on the hypothesis of retaining a model.&lt;br /&gt;
If it is clear from the beginning that for example the structural model is totally misspecified (e.g., we use a linear function of time even though a curvature is clearly visible in the data), any basic goodness-of-fit plot (individual fits, observation vs prediction, residuals, etc.) will detect this misspecification without any doubt and without the need to evaluate the probability of making a mistake.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
To put into practice  such a statistical test, we usually construct a test statistic $T(\by)$ which is a function of the observations and for which we are able to calculate a distribution under the null hypothesis $H_0$.&lt;br /&gt;
&lt;br /&gt;
For a given significance level $\alpha$, we then define a rejection region $R_\alpha$ such that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation= &amp;lt;math&amp;gt;\probs{H_0}{T(\by) \in R_\alpha} = \alpha . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus, $\alpha$ is the probability of incorrectly rejecting the null hypothesis $H_0$.&lt;br /&gt;
&lt;br /&gt;
The difficulty in creating and using such tests comes from two main things:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* We need to be capable of calculating the distribution of the test statistic $T(\by)$ under $H_0$ in order to carefully track the significance level, i.e., ensure that the probability of incorrectly rejecting $H_0$ is indeed $\alpha$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* Being able to control the type I error $\alpha$ is of no interest if the test has low power, i.e., if the probability of correctly rejecting $H_0$ is low.&lt;br /&gt;
&amp;lt;/ul&amp;gt; &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In the present context, the first point is clearly a problem. Due to the complexity of the models we are interested in, it is impossible to analytically calculate the distribution of a function of the observations, even for something as simple as the empirical mean $\overline{\by}=\sum_{i,j}y_{ij}/\sum_i{n_i}$.&lt;br /&gt;
Using limit theorems to approximate such distributions is also more or less hopeless. Perhaps the most powerful, general and precise solution we have available to us is [http://en.wikipedia.org/wiki/Monte_Carlo_simulation Monte Carlo simulation]:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Generate independent values $\by^{(1)}, \by^{(2)}, \ldots , \by^{(L)}$ under the model ${\cal M}_0$ using the same design and covariates as in the original data. &lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* Calculate the $L$ statistics $T(\by^{(1)}), T(\by^{(2)}), \ldots , T(\by^{(L)})$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* Estimate the distribution of $T(\by)$ under ${\cal M}_0$ with the empirical distribution of the $T(\by^{(\ell)})$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The estimation error essentially depends on the number $L$ of  simulated data sets. We must therefore choose $L$ large enough so that this error is negligible.&lt;br /&gt;
&lt;br /&gt;
This solves part of the problem. But it remains hard to define a rejection region $R_\alpha$ if the test statistic is multidimensional. We can of course calculate  relevant prediction intervals for each component of $T(\by)$ individually, but this does not really help us to define a real multidimensional rejection region.&lt;br /&gt;
&lt;br /&gt;
Consequently, diagnostics methods are essentially visual: we compare the observed statistic $T(\by)$ with the expected distribution under ${\cal M}_0$ by graphically displaying (for example) 90% or 95% prediction intervals for each component of $T(\by)$.&lt;br /&gt;
&lt;br /&gt;
The second point is also delicate because we need to decide what ${\cal M}\neq{\cal M}_0$ means, i.e., $H_0$ being false.&lt;br /&gt;
Does it mean that the structural model is misspecified? Or the distribution of the random effects, the residual error model, the covariate model? There are so many ways in which a model can be misspecified that we cannot realistically expect to be able to create one unique statistic sufficiently powerful to detect all of these at once. We therefore prefer to construct several different test statistics, i.e., several graphical diagnostics tools, each good at dealing with one particular type of misspecification. It is then the combination of all these tools that will make up our test; we can fairly reasonably hope that a misspecified model will not succeed in passing through this filter.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Diagnostic plots using  individual parameters===&lt;br /&gt;
&lt;br /&gt;
Diagnostic plots constructed using only the observations are useful for looking at the distribution $\qy$ of the observations,&lt;br /&gt;
but do not help with testing hypotheses made on the non-observed individual parameters (about the distribution, covariate model, etc.).&lt;br /&gt;
&lt;br /&gt;
One possible solution is to estimate the individual parameters (using for example the conditional mode) and then use these estimates to create new diagnostic tools. This strategy is only useful when the individual parameters have been estimated well.&lt;br /&gt;
&lt;br /&gt;
If instead the data does not contain enough information to estimate certain individual parameters well, the individual estimates are all shrunk towards the same (population) value; this is the mode (resp. mean) of the population distribution of the parameter if we use the conditional mode (resp. conditional mean). For a parameter $\psi_i$ which is a function of a random effect $\eta_i$, we can quantify this phenomena  by defining the so-called $\eta$-shrinkage as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \eta\text{-shrinkage} = 1 - \displaystyle{\frac{\var{\hat{\eta}_i} }{\var{\eta_i} } }, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{\eta}_i$ is an estimate of $\eta_i$ (the conditional mode, conditional mean, etc.).&lt;br /&gt;
&lt;br /&gt;
This shrinkage phenomenon is simple to understand because the conditional distribution $\qcetaiyi$ of $\eta_i$ is defined by the product $\pmacro(y_i|\eta_i)\pmacro(\eta_i)$. Saying that the observations $y_i$ provides little information about $\eta_i$ means that the conditional distribution of $y_i$ has a reduced importance in the construction of $\qcetaiyi$. The mode (resp. mean) of $\qcetaiyi$ will therefore be close to 0 which is both the mode and mean of $\qetai$. The result is a high level of shrinkage (close to 1) whenever $\var{\hat{\eta}_i}\ll\var{\eta_i}$.&lt;br /&gt;
&lt;br /&gt;
Estimates of the $\psi_i$ are therefore biased because they do not correctly reflect the marginal distribution $\qpsii$ (in particular, their variance is much less). A particularly effective solution is to simulate the individual parameters $\psi_i$ with the conditional distribution $\qcpsiiyi$ rather than taking the mode. The resulting estimator is unbiased in the following sense:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pmacro(\psi_i) &amp;amp;=&amp;amp; \displaystyle{ \int \pmacro(\psi_i {{!}}  y_i )\pmacro( y_i ) d\, y_i }\\&lt;br /&gt;
&amp;amp;=&amp;amp; \esps{y_i}{\pmacro(\psi_i {{!}} y_i )} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This relationship is a fundamental one when considering inverse problems, incomplete data models, mixed-effects models, etc. So what does it imply exactly? Well, if we randomly draw a vector $y_i$ of observations for an individual in a population and then generate a vector $\psi_i$ using the conditional distribution $\qcpsiiyi$, then the distribution of $\psi_i$ is the population distribution $\qpsii$. In other words, even if each $\psi_i$ is simulated using its own conditional distribution, the fact of pooling them allows us to look at them as if they were a sample from $\qpsii$, i.e., the marginal distribution $\qpsii$ is a mixture of conditional distributions $\qcpsiiyi$.&lt;br /&gt;
&lt;br /&gt;
The procedure is therefore as follows: we generate several values from each conditional distribution $\qcpsiiyi$ using the [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings algorithm]], and use them in addition to the observations in order to build various diagnostic plots.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== An example ===&lt;br /&gt;
&lt;br /&gt;
We are going to use the same model used for model exploration in the [[Visualization]] section and parameter estimation in the [[Estimation#Maximum likelihood estimation of the population parameters | Maximum likelihood estimation of the population parameters]] chapter.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model that defines the concentration in the central compartment and the hazard function for the events (hemorrhaging) is&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{D \, ka}{V(ka-Cl/V)} }\left(e^{-(Cl/V)\,t} - e^{-ka\,t} \right) \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The statistical model assumes that $ka_i$ and $V_i$ are log-normally distributed, $Cl_i$ normal, $h0_i$  probit-normal and $\gamma$ logit-normal. No covariates are used in this model. Lastly, we suppose a constant residual error model. Now we are going to review several different diagnostic plots and look at the conclusions that can be made using them.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
1) &amp;lt;u&amp;gt;Individual fits.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
In the continuous data model $y_{ij}=f(t_{ij};\psi_i) + f(t_{ij};\psi_i)\teps_{ij}$,  estimation of the population parameters $\psi_{\rm pop}$ and individual parameters $\psi_{i}$ allows us to compute for each individual:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $f(t ; \hat{\psi}_{\rm pop})$, the predicted profile given by the estimated population model&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $f(t ; \hat{\psi}_{i})$, the predicted profile given by the estimated individual model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text=The figure plotting for each individual the two curves for the predicted concentration shows evidence of inter-individual variability in the kinetics, and furthermore does not allow us to reject the proposed PK model since the fits seem acceptable.&lt;br /&gt;
|image=indfits1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2) &amp;lt;u&amp;gt;Observations vs predictions.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
The population and individual models also allow us to calculate for each individual predictions at the observation times&lt;br /&gt;
$f(t_{ij} ; \hat{\psi}_{\rm pop})$ and $f(t_{ij} ; \hat{\psi}_{i})$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The figure showing observations vs individuals reveals no obvious misspecification. In particular it does not allow us to reject the constant residual error model.&lt;br /&gt;
|image=obspred1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
3) &amp;lt;u&amp;gt;Residuals.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
Several types of residuals can be defined: ''population weighted residuals'' $({\rm PWRES}_{ij})$, ''individual weighted residuals'' $({\rm IWRES}_{ij})$, ''normalised prediction distribution errors'' $({\rm NPDE}_{ij})$, etc. The first two are defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\rm PWRES}_{ij} &amp;amp;=&amp;amp; \frac{y_{ij} - \hat{\mathbb{E} }(y_{ij})}{\hat{\rm std}(y_{ij})} \\ {\rm IWRES}_{ij} &amp;amp;=&amp;amp; \frac{y_{ij} - f(t_{ij} ; \hat{\psi}_{i})}{g(t_{ij} ; \hat{\psi}_{i})},&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{\mathbb{E}}(y_{ij})$ and $\hat{\rm std}(y_{ij})$ are the  mean and variance of $y_{ij}$ estimated by Monte Carlo.&lt;br /&gt;
${\rm NPDE}_{ij}$ is a nonparametric version of ${\rm PWRES}_{ij}$, based on a rank statistic. See [http://www.npde.biostat.fr  npde] for more details.&lt;br /&gt;
&lt;br /&gt;
Statistics useful for summarizing the residuals are the  10%, 50% (median) and 90% quantiles. The procedure described earlier for estimating prediction intervals of these quantiles using Monte Carlo can again be used.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= Both the individual residuals and the NPDEs suggest that the model is misspecified. Indeed, under ${\cal M}_0$ residuals are expected to behave as i.i.d. ${\cal N}(0,1)$ random variables, which is clearly not the case here. It is nevertheless difficult to identify the reasons for this misspecification using only these figures.&lt;br /&gt;
|image=residual1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= These plots show at each observation time the order 10%, 50% and 90% quantiles of the IWRES and NPDE.&lt;br /&gt;
The 90% prediction intervals are also displayed. These plots are more informative than the original residual plots.&lt;br /&gt;
We can now reasonably conclude that the behavior of the three quantiles is not the one expected under ${\cal M}_0$. In particular, a proportional component in the residual error model appears not to have been taken into account.&lt;br /&gt;
|image=residual2.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
4) &amp;lt;u&amp;gt;The distributions of the individual parameters.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
The hypotheses we have made about the distributions of the individual parameters can be tested by visually comparing the pdf of the pre-selected distribution of each parameter with the empirical distribution of that parameter. We are going to see that using the estimated individual parameters does not allow us to construct a pertinent diagnostic plot, and that we must rather use parameters simulated with the conditional distribution  $\qcpsiiyi$ for each individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The plots show for each model parameter the pdf obtained from the estimated parameters and the empirical distribution, shown as a histogram, of the estimated individual parameters (here the estimated parameters are the modes of the conditional distributions).&lt;br /&gt;
|image=distparam0.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= Instead of histograms, $\monolix$ can also display the empirical distribution of a continuous variable using  nonparametric density estimation. This is a better way to represent continuous distributions than a histogram.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
It is also possible to display the $\eta$-shrinkage for each parameter. As expected, $\eta$-shrinkage is large for the parameters associated with the time-to-events process. That does not mean that the statistical models for $h_0$ and $\gamma$ are misspecified, but that the data does not allow us to correctly recover these individual parameters.&lt;br /&gt;
|image=distparam1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The simulated individual parameters now allow us to construct a diagnostic tool for the distributions of the individual parameters. Only the distribution of the clearance $Cl$ would appear to be rejected. Though the model had supposed a normal distribution, the simulated parameters seem to suggest an asymmetric distribution. This diagnostic plot leads us to think about testing the hypothesis of a log-normal distribution for $Cl$.&lt;br /&gt;
|image=distparam2.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
5) &amp;lt;u&amp;gt;Covariate model.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
The model assumes that for a given individual parameter $\psi_i$, there exists a function $h$ such that $h(\psi_i) = h(\psi_{\rm pop}) + \eta_i$, where the random effects are i.i.d. Gaussian variables. We can then graphically display the random effects simulated with the conditional distributions as a function of the various covariates in order to see whether this hypothesis is valid.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= These plots clearly show that the simulated random effects for $V$ and $Cl$ are correlated with the weight and have different distributions depending on gender. The assumption that volume $V$ and clearance $Cl$ are independent of weight should be rejected. The statistical model also needs to take into account the fact that both predicted volume and clearance increase with weight.&lt;br /&gt;
|image=Eval_covariate2.png &lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
6) &amp;lt;u&amp;gt;The correlation model.&amp;lt;/u&amp;gt;&lt;br /&gt;
The model assumes that for a given individual, the random effects associated with the each individual parameter are independent.&lt;br /&gt;
We can plot each pair of  random effects simulated with the conditional distributions against each other to see if this hypothesis is valid.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text=The various point clouds  show no correlation between random effects except for $(\eta_V,\eta_{Cl})$ and perhaps $(\eta_{ka},\eta_V)$.&lt;br /&gt;
|image=correl1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
7) &amp;lt;u&amp;gt;Visual predictive checks.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
A VPC is a diagnostic tool well suited to continuous data. It allows us to summarize in the same figure the structural and statistical models. The VPC shown uses the order 10%, 50% and 90% for the observations after having regrouped them into bins for successive intervals. Then, prediction intervals for these quantiles under ${\cal M}_0$ are estimated using Monte Carlo.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text=To make it easier to interpret VPCs, we represent in red the zones where the observed quantiles are outside the prediction intervals. Here, the structural model seems to be ok, but the statistical model exhibits some incoherencies. In particular, the three quantiles obtained using the observations appear much closer together than the model ${\cal M}_0$ would suggest. This adds weight to the suggestion that a proportional component should be added to the error model.&lt;br /&gt;
|image=vpc4.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
8) &amp;lt;u&amp;gt;Kaplan Meier plots.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
Special diagnostic plots need to be defined for non-continuous observations.&lt;br /&gt;
We can for survival (time-to-event) data use  Kaplan Meier plots (for the first event) as a statistic, and/or the average cumulated number of events per individual (i.e., the mean number of observed events before time $t$). The prediction intervals for these statistics can be estimated by Monte Carlo.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The shapes of the curves seem correct. The model appears to slightly overestimate the survival function after the 15 hr mark, but it is difficult at this point to decide whether this comes from the time-to-event model itself,  the statistical model or  the model for the concentration.&lt;br /&gt;
|image=km2.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=  In summary, this ensemble of diagnostic plots has suggested to us that we should suppose:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* a log-normal distribution for $Cl$&lt;br /&gt;
&lt;br /&gt;
* a combined residual error model, e.g., $y=f+(a+b*f)\teps$&lt;br /&gt;
&lt;br /&gt;
* a statistical model for $Cl$ and $V$ which incorporates weight as a covariate, assuming for instance a linear relationship between $\log(V)$ and $\log({\rm weight})$ and a linear relationship between $\log(Cl)$ and $\log({\rm weight})$&lt;br /&gt;
&lt;br /&gt;
*a linear correlation between $\log(Cl)$ and $\log(V)$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This new model can  be easily implemented in $\monolix$. Population parameters and individual parameters are then estimated anew and new diagnostic plots drawn. A few of these are displayed below and clearly show that the new model is better than the previous one, and can likely be retained as a candidate model.&lt;br /&gt;
&lt;br /&gt;
::[[File:diagnostic.png]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model selection == &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Statistical tools for model selection ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Statistical tools for model selection include information criteria (AIC and BIC) and hypothesis tests such as the Wald test and  likelihood ratio test (LRT).&lt;br /&gt;
&lt;br /&gt;
The Akaike Information Criteria (AIC) and the Bayesian information Criteria (BIC) are defined by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
AIC &amp;amp;=&amp;amp; - 2 {\llike}(\theta;\by) + 2 P \\&lt;br /&gt;
BIC &amp;amp;=&amp;amp; - 2 {\llike}(\theta;\by) + \log(N) P ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $P$ is the total number of parameters to be estimated and $N$ the number of subjects. The models being compared using AIC or BIC need not be nested, unlike the case when models are being compared using an F-test or likelihood ratio test.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=  	&amp;amp;#32;&lt;br /&gt;
* Surprisingly, the formula for calculating the BIC  differs from one software to another. The reason is that the effective sample size is not clearly defined in the context of mixed-effects models. The question is whether we should use the number of subjects $N$ or the total number of observations $n_{\mathrm{tot} }$ in the penalty term. The penalty using $n_{\mathrm{tot} }$ is implemented in the R package nlme &amp;lt;!--[http://cran.r-project.org/web/packages/nlme/nlme.pdf  nlme]--&amp;gt; and  the SPSS procedure MIXED &amp;lt;!--[http://www.spss.ch/upload/1126184451_Linear%20Mixed%20Effects%20Modeling%20in%20SPSS.pdf   MIXED]--&amp;gt;, while $N$  is used in saemix for $\monolix$, &amp;lt;!--[http://cran.r-project.org/web/packages/saemix/saemix.pdf  saemix]--&amp;gt; and in SAS proc NLMIXED&amp;lt;!--[http://support.sas.com/documentation/cdl/en/statug/63033/HTML/default/viewer.htm#nlmixed_toc.htm  NLMIXED]--&amp;gt;.&lt;br /&gt;
&lt;br /&gt;
: An appropriate decomposition of the complete log-likelihood combined with the Laplace approximation can be used to derive the asymptotic BIC approximation. This leads to an optimal BIC penalty based on two terms proportional to $\log N$ and $\log n_{\mathrm{tot} }$ that adapts to the mixed-effects structure of the model. This new approach is not implemented yet in any software.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* AIC and BIC are justified based on asymptotic criteria (the AIC heuristic uses Wilks' theorem and BIC uses Bayesian statistics), i.e., when the number of individuals increases and the model dimension stays fixed. In the alternative, non-asymptotic approach, the model size can increase freely. The form of the penalty can differ from one model to the next in this framework. It can be shown for example that for certain Gaussian models, the penalty term has the form   $c_1P + c_2P\log(N/P)$. The problem then becomes to calibrate the coefficients $c_1$ and $c_2$ in order to obtain an optimal penalty, which is not necessarily a simple task, making it harder to use this approach in real applications.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The observed log-likelihood ${\llike}(\theta;\by) = \log(\py(\by;\theta))$ cannot be computed in closed form for nonlinear mixed-effects models. It can be estimated by Monte Carlo using the importance sampling algorithm described in the [[Estimation of the log-likelihood]] section.&lt;br /&gt;
&lt;br /&gt;
When comparing two nested models ${\cal M}_0$ and ${\cal M}_1$ with dimensions $P_0$ and $P_1$ (with $P_1&amp;gt;P_0$), the ''likelihood ratio test'' (LRT) uses the test statistic&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
LRT = 2 ( {\llike}(\hthetag_1;\by) -  {\llike}(\hthetag_0;\by) ) ,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hthetag_0$ and $\hthetag_1$ are the MLE of $\theta$ under ${\cal M}_0$ and ${\cal M}_1$.&lt;br /&gt;
&lt;br /&gt;
Depending on the hypotheses, the limit distribution of $LRT$ is either a $\chi ^2$ distribution or a mixture of a $\chi^2$ distribution and a Dirac $\delta$ distribution. For example:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
-  Testing whether some fixed effects are null and assuming the same covariance structure for the random effects implies that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; LRT \limite{N\to \infty}{} \chi^2(P_1-P_0) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
-  Testing whether some correlations in the covariance matrix $\IIV$ are null and assuming the same covariate model implies that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
LRT \limite{N\to \infty}{} \chi^2(P_1-P_0) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
- Testing whether the variance of one of the random effects is zero and assuming the same covariate model  implies that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
LRT \limite{N\to \infty}{} \displaystyle{ \frac{1}{2} }\chi^2(1) + \displaystyle{ \frac{1}{2} }\delta_0 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Statistical tests,  as is the case for BIC, help us decide whether the difference between two models is statistically significant.&lt;br /&gt;
Suppose that we want to test  whether a fixed model parameter $\beta$ is null:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
H_0: \quad \beta=0 \quad vs \quad H_1: \quad \beta\neq0 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
We construct a statistical test $T$ for which the distribution under $H_0$ allows us to calculate a $p$-value, i.e., the probability that $T$ is at least as big as the value observed under $H_0$. A small $p$-value leads us to reject $H_0$ with high confidence. It is usual to use the arbitrary cutoff of 5% to make this decision: we frequently read statements such as &amp;quot;a decrease in the objective function of a least 3.84 was required to identify a signifiant covariate&amp;quot;.&lt;br /&gt;
In the same way we could select models based on their BIC values under $H_0$ and $H_1$ by providing an arbitrary decision rule. It is sometimes suggested to choose $H_1$ if the difference $BIC_{H_1} -BIC_{H_0}$ is inferior to a certain arbitrary cutoff.&lt;br /&gt;
&lt;br /&gt;
These approaches seem to simplify the modeler's life because they provide decision rules that can be applied systematically without thinking and thus justify decisions. But a rule, whatever it is, should never stop us asking why we are applying it and whether it is applicable in the present case. Remember that even a very small difference will be statistically significant if the sample size is large enough. The question is thus not  to know whether a difference is statistically significant, but whether it is physically or biologically significant. We must therefore look carefully at the size of an effect and its real impact, both for understanding the model and for understanding its impact on the model's predictive capacities.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===An example===&lt;br /&gt;
&lt;br /&gt;
We are going to continue using our joint model for concentration data and hemorrhaging events. The structural model is chosen to be the one used earlier for the diagnostic tests. We would now like to compare several statistical models:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_1$: all the individual parameters are log-normally distributed, there is no correlation between individual parameters,  the residual error model is a combined one ($y=f+(a+b*f)*\teps$), both $\log(V)$ and $\log(Cl)$ are linear functions of $\log({\rm weight})$.  &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_2$: model ${\cal M}_1$, assuming that $\log(V_i)$ and $\log(Cl_i)$ are linearly correlated.  &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_3$: model ${\cal M}_2$, assuming that $\gamma_i=\gamma_{\rm pop}$ (i.e., $\omega_\gamma = 0$). &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_4$: model ${\cal M}_2$, assuming that $Cl$ is not a function of the weight.  &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The results for these four models are as follows:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; cellpadding=&amp;quot;20&amp;quot; cellspacing=&amp;quot;20&amp;quot; style=&amp;quot;margin-left:20%; margin-right=20%; width:50%&amp;quot;&lt;br /&gt;
! Model || $-2\times {\llike}$ || BIC &lt;br /&gt;
|-  &lt;br /&gt;
| ${\cal M}_1$ || 1390 || 1451 &lt;br /&gt;
|-&lt;br /&gt;
|  ${\cal M}_2$ || 1370 || 1436 &lt;br /&gt;
|-&lt;br /&gt;
|  ${\cal M}_3$ || 1370 ||1432 &lt;br /&gt;
|-&lt;br /&gt;
| ${\cal M}_4$ || 1446 || 1477 &lt;br /&gt;
|}   &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Clearly models ${\cal M}_2$ and ${\cal M}_3$ are the best in terms of BIC. We can therefore envisage selecting a model that includes both a correlation between $\log(V)$ and $\log(Cl)$ and that supposes that the volume $V$ is a function of the weight.&lt;br /&gt;
&lt;br /&gt;
Rather than a purely statistical criteria (LRT or BIC) it is particularly the estimated value of the standard deviation of  $\gamma$ ($\hat{\omega}_\gamma = 0.002$) under the model ${\cal M}_2$ that would lead us to conclude that the inter-individual variability of $\gamma$ is negligible.&lt;br /&gt;
&lt;br /&gt;
It is also important to try and evaluate the impact that a bad decision could have. Retaining $\hat{\omega}_\gamma = 0.002$ would have little impact on predictions because $\gamma_i$ is log-normally distributed, representing a variability of around 0.2%. Models ${\cal M}_2$ and ${\cal M}_3$ are thus practically identical and there is no particular advantage of selecting one over the other.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{akaike1983information,&lt;br /&gt;
  title={Information measures and model selection},&lt;br /&gt;
  author={Akaike, H.},&lt;br /&gt;
  journal={Bulletin of the International Statistical Institute},&lt;br /&gt;
  volume={50},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={277-291},&lt;br /&gt;
  year={1983}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{arlot2010survey,&lt;br /&gt;
  title={A survey of cross-validation procedures for model selection},&lt;br /&gt;
  author={Arlot, S. and Celisse, A.},&lt;br /&gt;
  journal={Statistics Surveys},&lt;br /&gt;
  volume={4},&lt;br /&gt;
  pages={40 - 79},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={The author, under a Creative Commons Attribution License}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{berger1996intrinsic,&lt;br /&gt;
  title={The intrinsic Bayes factor for model selection and prediction},&lt;br /&gt;
  author={Berger, J.O. and Pericchi, L.R.},&lt;br /&gt;
  journal={Journal of the American Statistical Association},&lt;br /&gt;
  volume={91},&lt;br /&gt;
  number={433},&lt;br /&gt;
  pages={109-122},&lt;br /&gt;
  year={1996},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bergstrand2011prediction,&lt;br /&gt;
  title={Prediction-corrected visual predictive checks for diagnosing nonlinear mixed-effects models},&lt;br /&gt;
  author={Bergstrand, M. and Hooker, A.C. and Wallin, J.E. and Karlsson, M.O.},&lt;br /&gt;
  journal={The AAPS journal},&lt;br /&gt;
  volume={13},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={143-151},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{burnham2002model,&lt;br /&gt;
  title={Model selection and multi-model inference: a practical information-theoretic approach},&lt;br /&gt;
  author={Burnham, K.P. and Anderson, D.R.},&lt;br /&gt;
  year={2002},&lt;br /&gt;
  publisher={Springer Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{comets2010model,&lt;br /&gt;
  title={Model evaluation in nonlinear mixed effect models, with applications to pharmacokinetics},&lt;br /&gt;
  author={Comets,E. and Brendel,K.},&lt;br /&gt;
  journal={Journal de la Soci&amp;amp;eacute;t&amp;amp;eacute; Fran&amp;amp;ccedil;aise de Statistique},&lt;br /&gt;
  volume={151},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={106-128},&lt;br /&gt;
  year={2010}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{comets2008computing,&lt;br /&gt;
  title={Computing normalised prediction distribution errors to evaluate nonlinear mixed-effect models: the npde add-on package for R},&lt;br /&gt;
  author={Comets, E. and Brendel, K. and Mentr&amp;amp;eacute;, F.},&lt;br /&gt;
  journal={Computer methods and programs in biomedicine},&lt;br /&gt;
  volume={90},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={154-166},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Elsevier}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hausman1984specification,&lt;br /&gt;
  title={Specification tests for the multinomial logit model},&lt;br /&gt;
  author={Hausman, J. and McFadden, D.},&lt;br /&gt;
  journal={Econometrica: Journal of the Econometric Society},&lt;br /&gt;
  pages={1219-1240},&lt;br /&gt;
  year={1984},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{Hooker2009Xpose,&lt;br /&gt;
author = {Hooker, A. and Karlsson, M.O. and Jonsson, E.N.},&lt;br /&gt;
title = {Model diagnostic using XPOSE},&lt;br /&gt;
url = {http://xpose.sourceforge.net/generic\_chm/xpose.VPC.html},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@INPROCEEDINGS{KarlssonHolford2008Tutorial,&lt;br /&gt;
  author = {Karlsson, M.O. and Holford, N.},&lt;br /&gt;
  title = {A Tutorial on Visual Predictive Checks},&lt;br /&gt;
  booktitle = {PAGE 2008},&lt;br /&gt;
  year = {2008},&lt;br /&gt;
  owner = {kb},&lt;br /&gt;
  timestamp = {2011.04.18},&lt;br /&gt;
  url = {http://www.page-meeting.org/pdf\_assets/8694-Karlsson_Holford_VPC_Tutorial_hires.pdf}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{lavielle2011automatic,&lt;br /&gt;
  title={Automatic data binning for improved visual diagnosis of pharmacometric models},&lt;br /&gt;
  author={Lavielle, M. and Bleakley, K.},&lt;br /&gt;
  journal={Journal of pharmacokinetics and pharmacodynamics},&lt;br /&gt;
  volume={38},&lt;br /&gt;
  number={6},&lt;br /&gt;
  pages={861-871},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@misc{massart2007concentration,&lt;br /&gt;
  title={Concentration Inequalities and Model Selection, Ecole d’Et&amp;amp;eacute; de Probabilit&amp;amp;eacute;s de Saint-Flour XXXIII-2003 Lecture Notes in Mathematics 1896},&lt;br /&gt;
  author={Massart, P.},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Springer-Verlag, Berlin}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{neyman1992problem,&lt;br /&gt;
  title={On the problem of the most efficient tests of statistical hypotheses},&lt;br /&gt;
  author={Neyman, J. and Pearson, E.S.},&lt;br /&gt;
  year={1992},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{schwarz1978estimating,&lt;br /&gt;
  title={Estimating the dimension of a model},&lt;br /&gt;
  author={Schwarz, G.},&lt;br /&gt;
  journal={The annals of statistics},&lt;br /&gt;
  volume={6},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={461-464},&lt;br /&gt;
  year={1978},&lt;br /&gt;
  publisher={Institute of Mathematical Statistics}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{shimodaira1999multiple,&lt;br /&gt;
  title={Multiple comparisons of log-likelihoods with applications to phylogenetic inference},&lt;br /&gt;
  author={Shimodaira, H. and Hasegawa, M.},&lt;br /&gt;
  journal={Molecular biology and evolution},&lt;br /&gt;
  volume={16},&lt;br /&gt;
  pages={1114-1116},&lt;br /&gt;
  year={1999},&lt;br /&gt;
  publisher={UNIVERSITY OF CHICAGO PRESS}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{vaida2005conditional,&lt;br /&gt;
  title={Conditional Akaike information for mixed-effects models},&lt;br /&gt;
  author={Vaida, F. and Blanchard, S.},&lt;br /&gt;
  journal={Biometrika},&lt;br /&gt;
  volume={92},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={351--370},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Biometrika Trust}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wald1943tests,&lt;br /&gt;
  title={Tests of statistical hypotheses concerning several parameters when the number of observations is large},&lt;br /&gt;
  author={Wald, A.},&lt;br /&gt;
  journal={Transactions of the American Mathematical Society},&lt;br /&gt;
  volume={54},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={426-482},&lt;br /&gt;
  year={1943},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wilks1938large,&lt;br /&gt;
  title={The large-sample distribution of the likelihood ratio for testing composite hypotheses},&lt;br /&gt;
  author={Wilks, S.S.},&lt;br /&gt;
  journal={The Annals of Mathematical Statistics},&lt;br /&gt;
  volume={9},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={60-62},&lt;br /&gt;
  year={1938},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{zhao2006model,&lt;br /&gt;
  title={On model selection consistency of Lasso},&lt;br /&gt;
  author={Zhao, P. and Yu, B.},&lt;br /&gt;
  journal={The Journal of Machine Learning Research},&lt;br /&gt;
  volume={7},&lt;br /&gt;
  pages={2541-2563},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={JMLR. org}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Visualization&lt;br /&gt;
|linkNext=Simulation }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Model_evaluation&amp;diff=7453</id>
		<title>Model evaluation</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Model_evaluation&amp;diff=7453"/>
		<updated>2013-06-25T14:21:55Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Defining the expression &amp;quot;model evaluation&amp;quot; is a harder task than it may appear at first glance. Intuitively, it would seem to suggest evaluating the performance of a model based on the observed data, the same data that was used to build the model. Fair enough, but what then do we mean by the &amp;quot;performance&amp;quot; of a model?&lt;br /&gt;
&lt;br /&gt;
Do we mean the ability of the model to characterize and explain the phenomena being studied, in which case the goal is to use the model to understand the phenomena? Or do we mean the model's predictive performance when the model is used to predict the phenomena's behavior, either in the future or under new experimental conditions?&lt;br /&gt;
&lt;br /&gt;
What is comes down to is this: do we want to use the model to understand or to predict? This is the key question to ask before even starting to think about what tools to use and tasks to execute.&lt;br /&gt;
&lt;br /&gt;
Here, we will be for the most part focused on the ability of a model to explain the phenomena and data. Therefore, the first goal will be to check whether the data are in agreement with the model, and vice versa. In this process, model diagnostics can be used to eliminate model candidates that do not seem capable of reproducing the observed data.&lt;br /&gt;
As is the usual case in statistics, it is not because a model has not been rejected that it is necessarily the &amp;quot;true&amp;quot; one. All that we can say is that the experimental data does not allow us to reject this model. It is merely one of perhaps many models that cannot be rejected. Indeed, we can usually find several models that get past this first diagnostic step and are therefore not rejected.&lt;br /&gt;
&lt;br /&gt;
What to do, then, when several possible models are retained? Well, we can try to select the &amp;quot;best&amp;quot; one (or best ones if no leader distinguishes itself from the rest). This means developing a model selection process which allows us to compare the models to each other. Compare models? But with what criteria?&lt;br /&gt;
&lt;br /&gt;
In a purely explanatory context, [http://en.wikipedia.org/wiki/Occam%27s_razor Occam's razor] is a useful parsimony principle which states that among competing hypotheses, the one with the fewest assumptions should be selected. In a modeling context, this means that among valid competing  models, the most simple one should be selected.&lt;br /&gt;
&lt;br /&gt;
Model diagnostic tools are for the most part graphical or visual: we  &amp;quot;see&amp;quot; when something is not right between a chosen model and the data it is hypothesized to describe. Model selection tools, on the other hand, are analytical: we calculate the value of some criteria that allows us to compare the models with each other. However, it is absolutely critical to keep in mind the limits of these tools. These are not decision-making tools. It is not a $p$-value or some information criteria that can automatically decide which model to choose. It is always the modeler who must have the last word! This person uses the model diagnostic and selection tools in order to guide their decision, but at the end, it is they that must make the final decision. There is nothing more dangerous that rules applied without thinking and arbitrary cut-off values used blindly without reflection.&lt;br /&gt;
&lt;br /&gt;
Here, we will not look model selection techniques based on the various models' predictive performances. One such approach consists of splitting the data into three sets: a ''learning set'' is used for fitting the model, a ''validation set''  for choosing between models and a ''test set'' to assess the quality of the predictions made by the chosen model.&lt;br /&gt;
&lt;br /&gt;
Very few model diagnostic and selection tools exist for mixed-effects models. One of the most complete is [http://xpose.sourceforge.net Xpose], an R-based model-building aid for population analysis using [http://www.iconplc.com/technology/products/nonmem/ NONMEM]   that facilitates model diagnostics, candidate covariate identification and model comparison. Here we will use $\monolix$ and illustrate these techniques with the  example used previously for model exploration and parameter estimation.&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model diagnostics==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Model diagnostics  and statistical tests ===&lt;br /&gt;
&lt;br /&gt;
Suppose first that a model has been entirely defined by the modeler, and that its parameters have either been chosen or estimated.&lt;br /&gt;
What we call the &amp;quot;model&amp;quot; is therefore  a joint probability distribution along with some parameter values.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Note ${\cal M}_0$ the model we wish to evaluate. We place ourselves in the framework of statistical testing, and would like to perform the following hypothesis test:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; H_0: \quad {\cal M}={\cal M}_0 \quad vs \quad H_1: \quad {\cal M}\neq{\cal M}_0. &amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&amp;quot;Passing&amp;quot; the test does not mean that we accept $H_0$ but rather that we do not reject it. We will use the same point of view for model diagnostics whereby we eliminate model candidates that do not seem capable of reproducing the observed data, i.e., models for which we conclude that ${\cal M}\neq{\cal M}_0$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=Running a statistical test only makes sense when we have doubts on the hypothesis of retaining a model.&lt;br /&gt;
If it is clear from the beginning that for example the structural model is totally misspecified (e.g., we use a linear function of time even though a curvature is clearly visible in the data), any basic goodness-of-fit plot (individual fits, observation vs prediction, residuals, etc.) will detect this misspecification without any doubt and without the need to evaluate the probability of making a mistake.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
To put into practice  such a statistical test, we usually construct a test statistic $T(\by)$ which is a function of the observations and for which we are able to calculate a distribution under the null hypothesis $H_0$.&lt;br /&gt;
&lt;br /&gt;
For a given significance level $\alpha$, we then define a rejection region $R_\alpha$ such that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation= &amp;lt;math&amp;gt;\probs{H_0}{T(\by) \in R_\alpha} = \alpha . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus, $\alpha$ is the probability of incorrectly rejecting the null hypothesis $H_0$.&lt;br /&gt;
&lt;br /&gt;
The difficulty in creating and using such tests comes from two main things:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* We need to be capable of calculating the distribution of the test statistic $T(\by)$ under $H_0$ in order to carefully track the significance level, i.e., ensure that the probability of incorrectly rejecting $H_0$ is indeed $\alpha$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* Being able to control the type I error $\alpha$ is of no interest if the test has low power, i.e., if the probability of correctly rejecting $H_0$ is low.&lt;br /&gt;
&amp;lt;/ul&amp;gt; &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In the present context, the first point is clearly a problem. Due to the complexity of the models we are interested in, it is impossible to analytically calculate the distribution of a function of the observations, even for something as simple as the empirical mean $\overline{\by}=\sum_{i,j}y_{ij}/\sum_i{n_i}$.&lt;br /&gt;
Using limit theorems to approximate such distributions is also more or less hopeless. Perhaps the most powerful, general and precise solution we have available to us is Monte Carlo simulation:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Generate independent values $\by^{(1)}, \by^{(2)}, \ldots , \by^{(L)}$ under the model ${\cal M}_0$ using the same design and covariates as in the original data. &lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* Calculate the $L$ statistics $T(\by^{(1)}), T(\by^{(2)}), \ldots , T(\by^{(L)})$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* Estimate the distribution of $T(\by)$ under ${\cal M}_0$ with the empirical distribution of the $T(\by^{(\ell)})$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The estimation error essentially depends on the number $L$ of  simulated data sets. We must therefore choose $L$ large enough so that this error is negligible.&lt;br /&gt;
&lt;br /&gt;
This solves part of the problem. But it remains hard to define a rejection region $R_\alpha$ if the test statistic is multidimensional. We can of course calculate  relevant prediction intervals for each component of $T(\by)$ individually, but this does not really help us to define a real multidimensional rejection region.&lt;br /&gt;
&lt;br /&gt;
Consequently, diagnostics methods are essentially visual: we compare the observed statistic $T(\by)$ with the expected distribution under ${\cal M}_0$ by graphically displaying (for example) 90% or 95% prediction intervals for each component of $T(\by)$.&lt;br /&gt;
&lt;br /&gt;
The second point is also delicate because we need to decide what ${\cal M}\neq{\cal M}_0$ means, i.e., $H_0$ being false.&lt;br /&gt;
Does it mean that the structural model is misspecified? Or the distribution of the random effects, the residual error model, the covariate model? There are so many ways in which a model can be misspecified that we cannot realistically expect to be able to create one unique statistic sufficiently powerful to detect all of these at once. We therefore prefer to construct several different test statistics, i.e., several graphical diagnostics tools, each good at dealing with one particular type of misspecification. It is then the combination of all these tools that will make up our test; we can fairly reasonably hope that a misspecified model will not succeed in passing through this filter.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===Diagnostic plots using  individual parameters===&lt;br /&gt;
&lt;br /&gt;
Diagnostic plots constructed using only the observations are useful for looking at the distribution $\qy$ of the observations,&lt;br /&gt;
but do not help with testing hypotheses made on the non-observed individual parameters (about the distribution, covariate model, etc.).&lt;br /&gt;
&lt;br /&gt;
One possible solution is to estimate the individual parameters (using for example the conditional mode) and then use these estimates to create new diagnostic tools. This strategy is only useful when the individual parameters have been estimated well.&lt;br /&gt;
&lt;br /&gt;
If instead the data does not contain enough information to estimate certain individual parameters well, the individual estimates are all shrunk towards the same (population) value; this is the mode (resp. mean) of the population distribution of the parameter if we use the conditional mode (resp. conditional mean). For a parameter $\psi_i$ which is a function of a random effect $\eta_i$, we can quantify this phenomena  by defining the so-called $\eta$-shrinkage as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \eta\text{-shrinkage} = 1 - \displaystyle{\frac{\var{\hat{\eta}_i} }{\var{\eta_i} } }, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{\eta}_i$ is an estimate of $\eta_i$ (the conditional mode, conditional mean, etc.).&lt;br /&gt;
&lt;br /&gt;
This shrinkage phenomenon is simple to understand because the conditional distribution $\qcetaiyi$ of $\eta_i$ is defined by the product $\pmacro(y_i|\eta_i)\pmacro(\eta_i)$. Saying that the observations $y_i$ provides little information about $\eta_i$ means that the conditional distribution of $y_i$ has a reduced importance in the construction of $\qcetaiyi$. The mode (resp. mean) of $\qcetaiyi$ will therefore be close to 0 which is both the mode and mean of $\qetai$. The result is a high level of shrinkage (close to 1) whenever $\var{\hat{\eta}_i}\ll\var{\eta_i}$.&lt;br /&gt;
&lt;br /&gt;
Estimates of the $\psi_i$ are therefore biased because they do not correctly reflect the marginal distribution $\qpsii$ (in particular, their variance is much less). A particularly effective solution is to simulate the individual parameters $\psi_i$ with the conditional distribution $\qcpsiiyi$ rather than taking the mode. The resulting estimator is unbiased in the following sense:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pmacro(\psi_i) &amp;amp;=&amp;amp; \displaystyle{ \int \pmacro(\psi_i {{!}}  y_i )\pmacro( y_i ) d\, y_i }\\&lt;br /&gt;
&amp;amp;=&amp;amp; \esps{y_i}{\pmacro(\psi_i {{!}} y_i )} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This relationship is a fundamental one when considering inverse problems, incomplete data models, mixed-effects models, etc. So what does it imply exactly? Well, if we randomly draw a vector $y_i$ of observations for an individual in a population and then generate a vector $\psi_i$ using the conditional distribution $\qcpsiiyi$, then the distribution of $\psi_i$ is the population distribution $\qpsii$. In other words, even if each $\psi_i$ is simulated using its own conditional distribution, the fact of pooling them allows us to look at them as if they were a sample from $\qpsii$, i.e., the marginal distribution $\qpsii$ is a mixture of conditional distributions $\qcpsiiyi$.&lt;br /&gt;
&lt;br /&gt;
The procedure is therefore as follows: we generate several values from each conditional distribution $\qcpsiiyi$ using the [[The Metropolis-Hastings algorithm for simulating the individual parameters|Metropolis-Hastings algorithm]], and use them in addition to the observations in order to build various diagnostic plots.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== An example ===&lt;br /&gt;
&lt;br /&gt;
We are going to use the same model used for model exploration in the [[Visualization]] section and parameter estimation in the [[Estimation#Maximum likelihood estimation of the population parameters | Maximum likelihood estimation of the population parameters]] chapter.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The structural model that defines the concentration in the central compartment and the hazard function for the events (hemorrhaging) is&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{D \, ka}{V(ka-Cl/V)} }\left(e^{-(Cl/V)\,t} - e^{-ka\,t} \right) \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The statistical model assumes that $ka_i$ and $V_i$ are log-normally distributed, $Cl_i$ normal, $h0_i$  probit-normal and $\gamma$ logit-normal. No covariates are used in this model. Lastly, we suppose a constant residual error model. Now we are going to review several different diagnostic plots and look at the conclusions that can be made using them.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
1) &amp;lt;u&amp;gt;Individual fits.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
In the continuous data model $y_{ij}=f(t_{ij};\psi_i) + f(t_{ij};\psi_i)\teps_{ij}$,  estimation of the population parameters $\psi_{\rm pop}$ and individual parameters $\psi_{i}$ allows us to compute for each individual:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* $f(t ; \hat{\psi}_{\rm pop})$, the predicted profile given by the estimated population model&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $f(t ; \hat{\psi}_{i})$, the predicted profile given by the estimated individual model.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text=The figure plotting for each individual the two curves for the predicted concentration shows evidence of inter-individual variability in the kinetics, and furthermore does not allow us to reject the proposed PK model since the fits seem acceptable.&lt;br /&gt;
|image=indfits1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2) &amp;lt;u&amp;gt;Observations vs predictions.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
The population and individual models also allow us to calculate for each individual predictions at the observation times&lt;br /&gt;
$f(t_{ij} ; \hat{\psi}_{\rm pop})$ and $f(t_{ij} ; \hat{\psi}_{i})$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The figure showing observations vs individuals reveals no obvious misspecification. In particular it does not allow us to reject the constant residual error model.&lt;br /&gt;
|image=obspred1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
3) &amp;lt;u&amp;gt;Residuals.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
Several types of residuals can be defined: ''population weighted residuals'' $({\rm PWRES}_{ij})$, ''individual weighted residuals'' $({\rm IWRES}_{ij})$, ''normalised prediction distribution errors'' $({\rm NPDE}_{ij})$, etc. The first two are defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\rm PWRES}_{ij} &amp;amp;=&amp;amp; \frac{y_{ij} - \hat{\mathbb{E} }(y_{ij})}{\hat{\rm std}(y_{ij})} \\ {\rm IWRES}_{ij} &amp;amp;=&amp;amp; \frac{y_{ij} - f(t_{ij} ; \hat{\psi}_{i})}{g(t_{ij} ; \hat{\psi}_{i})},&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hat{\mathbb{E}}(y_{ij})$ and $\hat{\rm std}(y_{ij})$ are the  mean and variance of $y_{ij}$ estimated by Monte Carlo.&lt;br /&gt;
${\rm NPDE}_{ij}$ is a nonparametric version of ${\rm PWRES}_{ij}$, based on a rank statistic. See [http://www.npde.biostat.fr  npde] for more details.&lt;br /&gt;
&lt;br /&gt;
Statistics useful for summarizing the residuals are the  10%, 50% (median) and 90% quantiles. The procedure described earlier for estimating prediction intervals of these quantiles using Monte Carlo can again be used.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= Both the individual residuals and the NPDEs suggest that the model is misspecified. Indeed, under ${\cal M}_0$ residuals are expected to behave as i.i.d. ${\cal N}(0,1)$ random variables, which is clearly not the case here. It is nevertheless difficult to identify the reasons for this misspecification using only these figures.&lt;br /&gt;
|image=residual1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= These plots show at each observation time the order 10%, 50% and 90% quantiles of the IWRES and NPDE.&lt;br /&gt;
The 90% prediction intervals are also displayed. These plots are more informative than the original residual plots.&lt;br /&gt;
We can now reasonably conclude that the behavior of the three quantiles is not the one expected under ${\cal M}_0$. In particular, a proportional component in the residual error model appears not to have been taken into account.&lt;br /&gt;
|image=residual2.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
4) &amp;lt;u&amp;gt;The distributions of the individual parameters.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
The hypotheses we have made about the distributions of the individual parameters can be tested by visually comparing the pdf of the pre-selected distribution of each parameter with the empirical distribution of that parameter. We are going to see that using the estimated individual parameters does not allow us to construct a pertinent diagnostic plot, and that we must rather use parameters simulated with the conditional distribution  $\qcpsiiyi$ for each individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The plots show for each model parameter the pdf obtained from the estimated parameters and the empirical distribution, shown as a histogram, of the estimated individual parameters (here the estimated parameters are the modes of the conditional distributions).&lt;br /&gt;
|image=distparam0.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= Instead of histograms, $\monolix$ can also display the empirical distribution of a continuous variable using  nonparametric density estimation. This is a better way to represent continuous distributions than a histogram.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
It is also possible to display the $\eta$-shrinkage for each parameter. As expected, $\eta$-shrinkage is large for the parameters associated with the time-to-events process. That does not mean that the statistical models for $h_0$ and $\gamma$ are misspecified, but that the data does not allow us to correctly recover these individual parameters.&lt;br /&gt;
|image=distparam1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The simulated individual parameters now allow us to construct a diagnostic tool for the distributions of the individual parameters. Only the distribution of the clearance $Cl$ would appear to be rejected. Though the model had supposed a normal distribution, the simulated parameters seem to suggest an asymmetric distribution. This diagnostic plot leads us to think about testing the hypothesis of a log-normal distribution for $Cl$.&lt;br /&gt;
|image=distparam2.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
5) &amp;lt;u&amp;gt;Covariate model.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
The model assumes that for a given individual parameter $\psi_i$, there exists a function $h$ such that $h(\psi_i) = h(\psi_{\rm pop}) + \eta_i$, where the random effects are i.i.d. Gaussian variables. We can then graphically display the random effects simulated with the conditional distributions as a function of the various covariates in order to see whether this hypothesis is valid.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= These plots clearly show that the simulated random effects for $V$ and $Cl$ are correlated with the weight and have different distributions depending on gender. The assumption that volume $V$ and clearance $Cl$ are independent of weight should be rejected. The statistical model also needs to take into account the fact that both predicted volume and clearance increase with weight.&lt;br /&gt;
|image=Eval_covariate2.png &lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
6) &amp;lt;u&amp;gt;The correlation model.&amp;lt;/u&amp;gt;&lt;br /&gt;
The model assumes that for a given individual, the random effects associated with the each individual parameter are independent.&lt;br /&gt;
We can plot each pair of  random effects simulated with the conditional distributions against each other to see if this hypothesis is valid.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text=The various point clouds  show no correlation between random effects except for $(\eta_V,\eta_{Cl})$ and perhaps $(\eta_{ka},\eta_V)$.&lt;br /&gt;
|image=correl1.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
7) &amp;lt;u&amp;gt;Visual predictive checks.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
A VPC is a diagnostic tool well suited to continuous data. It allows us to summarize in the same figure the structural and statistical models. The VPC shown uses the order 10%, 50% and 90% for the observations after having regrouped them into bins for successive intervals. Then, prediction intervals for these quantiles under ${\cal M}_0$ are estimated using Monte Carlo.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text=To make it easier to interpret VPCs, we represent in red the zones where the observed quantiles are outside the prediction intervals. Here, the structural model seems to be ok, but the statistical model exhibits some incoherencies. In particular, the three quantiles obtained using the observations appear much closer together than the model ${\cal M}_0$ would suggest. This adds weight to the suggestion that a proportional component should be added to the error model.&lt;br /&gt;
|image=vpc4.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
8) &amp;lt;u&amp;gt;Kaplan Meier plots.&amp;lt;/u&amp;gt;&lt;br /&gt;
&lt;br /&gt;
Special diagnostic plots need to be defined for non-continuous observations.&lt;br /&gt;
We can for survival (time-to-event) data use  Kaplan Meier plots (for the first event) as a statistic, and/or the average cumulated number of events per individual (i.e., the mean number of observed events before time $t$). The prediction intervals for these statistics can be estimated by Monte Carlo.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithImage&lt;br /&gt;
|text= The shapes of the curves seem correct. The model appears to slightly overestimate the survival function after the 15 hr mark, but it is difficult at this point to decide whether this comes from the time-to-event model itself,  the statistical model or  the model for the concentration.&lt;br /&gt;
|image=km2.png&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{OutlineText&lt;br /&gt;
|text=  In summary, this ensemble of diagnostic plots has suggested to us that we should suppose:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* a log-normal distribution for $Cl$&lt;br /&gt;
&lt;br /&gt;
* a combined residual error model, e.g., $y=f+(a+b*f)\teps$&lt;br /&gt;
&lt;br /&gt;
* a statistical model for $Cl$ and $V$ which incorporates weight as a covariate, assuming for instance a linear relationship between $\log(V)$ and $\log({\rm weight})$ and a linear relationship between $\log(Cl)$ and $\log({\rm weight})$&lt;br /&gt;
&lt;br /&gt;
*a linear correlation between $\log(Cl)$ and $\log(V)$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This new model can  be easily implemented in $\monolix$. Population parameters and individual parameters are then estimated anew and new diagnostic plots drawn. A few of these are displayed below and clearly show that the new model is better than the previous one, and can likely be retained as a candidate model.&lt;br /&gt;
&lt;br /&gt;
::[[File:diagnostic.png]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Model selection == &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Statistical tools for model selection ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Statistical tools for model selection include information criteria (AIC and BIC) and hypothesis tests such as the Wald test and  likelihood ratio test (LRT).&lt;br /&gt;
&lt;br /&gt;
The Akaike Information Criteria (AIC) and the Bayesian information Criteria (BIC) are defined by&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
AIC &amp;amp;=&amp;amp; - 2 {\llike}(\theta;\by) + 2 P \\&lt;br /&gt;
BIC &amp;amp;=&amp;amp; - 2 {\llike}(\theta;\by) + \log(N) P ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $P$ is the total number of parameters to be estimated and $N$ the number of subjects. The models being compared using AIC or BIC need not be nested, unlike the case when models are being compared using an F-test or likelihood ratio test.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=  	&amp;amp;#32;&lt;br /&gt;
* Surprisingly, the formula for calculating the BIC  differs from one software to another. The reason is that the effective sample size is not clearly defined in the context of mixed-effects models. The question is whether we should use the number of subjects $N$ or the total number of observations $n_{\mathrm{tot} }$ in the penalty term. The penalty using $n_{\mathrm{tot} }$ is implemented in the R package nlme &amp;lt;!--[http://cran.r-project.org/web/packages/nlme/nlme.pdf  nlme]--&amp;gt; and  the SPSS procedure MIXED &amp;lt;!--[http://www.spss.ch/upload/1126184451_Linear%20Mixed%20Effects%20Modeling%20in%20SPSS.pdf   MIXED]--&amp;gt;, while $N$  is used in saemix for $\monolix$, &amp;lt;!--[http://cran.r-project.org/web/packages/saemix/saemix.pdf  saemix]--&amp;gt; and in SAS proc NLMIXED&amp;lt;!--[http://support.sas.com/documentation/cdl/en/statug/63033/HTML/default/viewer.htm#nlmixed_toc.htm  NLMIXED]--&amp;gt;.&lt;br /&gt;
&lt;br /&gt;
: An appropriate decomposition of the complete log-likelihood combined with the Laplace approximation can be used to derive the asymptotic BIC approximation. This leads to an optimal BIC penalty based on two terms proportional to $\log N$ and $\log n_{\mathrm{tot} }$ that adapts to the mixed-effects structure of the model. This new approach is not implemented yet in any software.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* AIC and BIC are justified based on asymptotic criteria (the AIC heuristic uses Wilks' theorem and BIC uses Bayesian statistics), i.e., when the number of individuals increases and the model dimension stays fixed. In the alternative, non-asymptotic approach, the model size can increase freely. The form of the penalty can differ from one model to the next in this framework. It can be shown for example that for certain Gaussian models, the penalty term has the form   $c_1P + c_2P\log(N/P)$. The problem then becomes to calibrate the coefficients $c_1$ and $c_2$ in order to obtain an optimal penalty, which is not necessarily a simple task, making it harder to use this approach in real applications.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The observed log-likelihood ${\llike}(\theta;\by) = \log(\py(\by;\theta))$ cannot be computed in closed form for nonlinear mixed-effects models. It can be estimated by Monte Carlo using the importance sampling algorithm described in the [[Estimation of the log-likelihood]] section.&lt;br /&gt;
&lt;br /&gt;
When comparing two nested models ${\cal M}_0$ and ${\cal M}_1$ with dimensions $P_0$ and $P_1$ (with $P_1&amp;gt;P_0$), the ''likelihood ratio test'' (LRT) uses the test statistic&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
LRT = 2 ( {\llike}(\hthetag_1;\by) -  {\llike}(\hthetag_0;\by) ) ,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\hthetag_0$ and $\hthetag_1$ are the MLE of $\theta$ under ${\cal M}_0$ and ${\cal M}_1$.&lt;br /&gt;
&lt;br /&gt;
Depending on the hypotheses, the limit distribution of $LRT$ is either a $\chi ^2$ distribution or a mixture of a $\chi^2$ distribution and a Dirac $\delta$ distribution. For example:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
-  Testing whether some fixed effects are null and assuming the same covariance structure for the random effects implies that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; LRT \limite{N\to \infty}{} \chi^2(P_1-P_0) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
-  Testing whether some correlations in the covariance matrix $\IIV$ are null and assuming the same covariate model implies that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
LRT \limite{N\to \infty}{} \chi^2(P_1-P_0) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
- Testing whether the variance of one of the random effects is zero and assuming the same covariate model  implies that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
LRT \limite{N\to \infty}{} \displaystyle{ \frac{1}{2} }\chi^2(1) + \displaystyle{ \frac{1}{2} }\delta_0 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Statistical tests,  as is the case for BIC, help us decide whether the difference between two models is statistically significant.&lt;br /&gt;
Suppose that we want to test  whether a fixed model parameter $\beta$ is null:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
H_0: \quad \beta=0 \quad vs \quad H_1: \quad \beta\neq0 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
We construct a statistical test $T$ for which the distribution under $H_0$ allows us to calculate a $p$-value, i.e., the probability that $T$ is at least as big as the value observed under $H_0$. A small $p$-value leads us to reject $H_0$ with high confidence. It is usual to use the arbitrary cutoff of 5% to make this decision: we frequently read statements such as &amp;quot;a decrease in the objective function of a least 3.84 was required to identify a signifiant covariate&amp;quot;.&lt;br /&gt;
In the same way we could select models based on their BIC values under $H_0$ and $H_1$ by providing an arbitrary decision rule. It is sometimes suggested to choose $H_1$ if the difference $BIC_{H_1} -BIC_{H_0}$ is inferior to a certain arbitrary cutoff.&lt;br /&gt;
&lt;br /&gt;
These approaches seem to simplify the modeler's life because they provide decision rules that can be applied systematically without thinking and thus justify decisions. But a rule, whatever it is, should never stop us asking why we are applying it and whether it is applicable in the present case. Remember that even a very small difference will be statistically significant if the sample size is large enough. The question is thus not  to know whether a difference is statistically significant, but whether it is physically or biologically significant. We must therefore look carefully at the size of an effect and its real impact, both for understanding the model and for understanding its impact on the model's predictive capacities.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
===An example===&lt;br /&gt;
&lt;br /&gt;
We are going to continue using our joint model for concentration data and hemorrhaging events. The structural model is chosen to be the one used earlier for the diagnostic tests. We would now like to compare several statistical models:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_1$: all the individual parameters are log-normally distributed, there is no correlation between individual parameters,  the residual error model is a combined one ($y=f+(a+b*f)*\teps$), both $\log(V)$ and $\log(Cl)$ are linear functions of $\log({\rm weight})$.  &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_2$: model ${\cal M}_1$, assuming that $\log(V_i)$ and $\log(Cl_i)$ are linearly correlated.  &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_3$: model ${\cal M}_2$, assuming that $\gamma_i=\gamma_{\rm pop}$ (i.e., $\omega_\gamma = 0$). &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
${\cal M}_4$: model ${\cal M}_2$, assuming that $Cl$ is not a function of the weight.  &lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The results for these four models are as follows:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; cellpadding=&amp;quot;20&amp;quot; cellspacing=&amp;quot;20&amp;quot; style=&amp;quot;margin-left:20%; margin-right=20%; width:50%&amp;quot;&lt;br /&gt;
! Model || $-2\times {\llike}$ || BIC &lt;br /&gt;
|-  &lt;br /&gt;
| ${\cal M}_1$ || 1390 || 1451 &lt;br /&gt;
|-&lt;br /&gt;
|  ${\cal M}_2$ || 1370 || 1436 &lt;br /&gt;
|-&lt;br /&gt;
|  ${\cal M}_3$ || 1370 ||1432 &lt;br /&gt;
|-&lt;br /&gt;
| ${\cal M}_4$ || 1446 || 1477 &lt;br /&gt;
|}   &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Clearly models ${\cal M}_2$ and ${\cal M}_3$ are the best in terms of BIC. We can therefore envisage selecting a model that includes both a correlation between $\log(V)$ and $\log(Cl)$ and that supposes that the volume $V$ is a function of the weight.&lt;br /&gt;
&lt;br /&gt;
Rather than a purely statistical criteria (LRT or BIC) it is particularly the estimated value of the standard deviation of  $\gamma$ ($\hat{\omega}_\gamma = 0.002$) under the model ${\cal M}_2$ that would lead us to conclude that the inter-individual variability of $\gamma$ is negligible.&lt;br /&gt;
&lt;br /&gt;
It is also important to try and evaluate the impact that a bad decision could have. Retaining $\hat{\omega}_\gamma = 0.002$ would have little impact on predictions because $\gamma_i$ is log-normally distributed, representing a variability of around 0.2%. Models ${\cal M}_2$ and ${\cal M}_3$ are thus practically identical and there is no particular advantage of selecting one over the other.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{akaike1983information,&lt;br /&gt;
  title={Information measures and model selection},&lt;br /&gt;
  author={Akaike, H.},&lt;br /&gt;
  journal={Bulletin of the International Statistical Institute},&lt;br /&gt;
  volume={50},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={277-291},&lt;br /&gt;
  year={1983}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{arlot2010survey,&lt;br /&gt;
  title={A survey of cross-validation procedures for model selection},&lt;br /&gt;
  author={Arlot, S. and Celisse, A.},&lt;br /&gt;
  journal={Statistics Surveys},&lt;br /&gt;
  volume={4},&lt;br /&gt;
  pages={40 - 79},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={The author, under a Creative Commons Attribution License}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{berger1996intrinsic,&lt;br /&gt;
  title={The intrinsic Bayes factor for model selection and prediction},&lt;br /&gt;
  author={Berger, J.O. and Pericchi, L.R.},&lt;br /&gt;
  journal={Journal of the American Statistical Association},&lt;br /&gt;
  volume={91},&lt;br /&gt;
  number={433},&lt;br /&gt;
  pages={109-122},&lt;br /&gt;
  year={1996},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bergstrand2011prediction,&lt;br /&gt;
  title={Prediction-corrected visual predictive checks for diagnosing nonlinear mixed-effects models},&lt;br /&gt;
  author={Bergstrand, M. and Hooker, A.C. and Wallin, J.E. and Karlsson, M.O.},&lt;br /&gt;
  journal={The AAPS journal},&lt;br /&gt;
  volume={13},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={143-151},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{burnham2002model,&lt;br /&gt;
  title={Model selection and multi-model inference: a practical information-theoretic approach},&lt;br /&gt;
  author={Burnham, K.P. and Anderson, D.R.},&lt;br /&gt;
  year={2002},&lt;br /&gt;
  publisher={Springer Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{comets2010model,&lt;br /&gt;
  title={Model evaluation in nonlinear mixed effect models, with applications to pharmacokinetics},&lt;br /&gt;
  author={Comets,E. and Brendel,K.},&lt;br /&gt;
  journal={Journal de la Soci&amp;amp;eacute;t&amp;amp;eacute; Fran&amp;amp;ccedil;aise de Statistique},&lt;br /&gt;
  volume={151},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={106-128},&lt;br /&gt;
  year={2010}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{comets2008computing,&lt;br /&gt;
  title={Computing normalised prediction distribution errors to evaluate nonlinear mixed-effect models: the npde add-on package for R},&lt;br /&gt;
  author={Comets, E. and Brendel, K. and Mentr&amp;amp;eacute;, F.},&lt;br /&gt;
  journal={Computer methods and programs in biomedicine},&lt;br /&gt;
  volume={90},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={154-166},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Elsevier}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hausman1984specification,&lt;br /&gt;
  title={Specification tests for the multinomial logit model},&lt;br /&gt;
  author={Hausman, J. and McFadden, D.},&lt;br /&gt;
  journal={Econometrica: Journal of the Econometric Society},&lt;br /&gt;
  pages={1219-1240},&lt;br /&gt;
  year={1984},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{Hooker2009Xpose,&lt;br /&gt;
author = {Hooker, A. and Karlsson, M.O. and Jonsson, E.N.},&lt;br /&gt;
title = {Model diagnostic using XPOSE},&lt;br /&gt;
url = {http://xpose.sourceforge.net/generic\_chm/xpose.VPC.html},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@INPROCEEDINGS{KarlssonHolford2008Tutorial,&lt;br /&gt;
  author = {Karlsson, M.O. and Holford, N.},&lt;br /&gt;
  title = {A Tutorial on Visual Predictive Checks},&lt;br /&gt;
  booktitle = {PAGE 2008},&lt;br /&gt;
  year = {2008},&lt;br /&gt;
  owner = {kb},&lt;br /&gt;
  timestamp = {2011.04.18},&lt;br /&gt;
  url = {http://www.page-meeting.org/pdf\_assets/8694-Karlsson_Holford_VPC_Tutorial_hires.pdf}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{lavielle2011automatic,&lt;br /&gt;
  title={Automatic data binning for improved visual diagnosis of pharmacometric models},&lt;br /&gt;
  author={Lavielle, M. and Bleakley, K.},&lt;br /&gt;
  journal={Journal of pharmacokinetics and pharmacodynamics},&lt;br /&gt;
  volume={38},&lt;br /&gt;
  number={6},&lt;br /&gt;
  pages={861-871},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@misc{massart2007concentration,&lt;br /&gt;
  title={Concentration Inequalities and Model Selection, Ecole d’Et&amp;amp;eacute; de Probabilit&amp;amp;eacute;s de Saint-Flour XXXIII-2003 Lecture Notes in Mathematics 1896},&lt;br /&gt;
  author={Massart, P.},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Springer-Verlag, Berlin}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{neyman1992problem,&lt;br /&gt;
  title={On the problem of the most efficient tests of statistical hypotheses},&lt;br /&gt;
  author={Neyman, J. and Pearson, E.S.},&lt;br /&gt;
  year={1992},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{schwarz1978estimating,&lt;br /&gt;
  title={Estimating the dimension of a model},&lt;br /&gt;
  author={Schwarz, G.},&lt;br /&gt;
  journal={The annals of statistics},&lt;br /&gt;
  volume={6},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={461-464},&lt;br /&gt;
  year={1978},&lt;br /&gt;
  publisher={Institute of Mathematical Statistics}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{shimodaira1999multiple,&lt;br /&gt;
  title={Multiple comparisons of log-likelihoods with applications to phylogenetic inference},&lt;br /&gt;
  author={Shimodaira, H. and Hasegawa, M.},&lt;br /&gt;
  journal={Molecular biology and evolution},&lt;br /&gt;
  volume={16},&lt;br /&gt;
  pages={1114-1116},&lt;br /&gt;
  year={1999},&lt;br /&gt;
  publisher={UNIVERSITY OF CHICAGO PRESS}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{vaida2005conditional,&lt;br /&gt;
  title={Conditional Akaike information for mixed-effects models},&lt;br /&gt;
  author={Vaida, F. and Blanchard, S.},&lt;br /&gt;
  journal={Biometrika},&lt;br /&gt;
  volume={92},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={351--370},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Biometrika Trust}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wald1943tests,&lt;br /&gt;
  title={Tests of statistical hypotheses concerning several parameters when the number of observations is large},&lt;br /&gt;
  author={Wald, A.},&lt;br /&gt;
  journal={Transactions of the American Mathematical Society},&lt;br /&gt;
  volume={54},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={426-482},&lt;br /&gt;
  year={1943},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wilks1938large,&lt;br /&gt;
  title={The large-sample distribution of the likelihood ratio for testing composite hypotheses},&lt;br /&gt;
  author={Wilks, S.S.},&lt;br /&gt;
  journal={The Annals of Mathematical Statistics},&lt;br /&gt;
  volume={9},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={60-62},&lt;br /&gt;
  year={1938},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{zhao2006model,&lt;br /&gt;
  title={On model selection consistency of Lasso},&lt;br /&gt;
  author={Zhao, P. and Yu, B.},&lt;br /&gt;
  journal={The Journal of Machine Learning Research},&lt;br /&gt;
  volume={7},&lt;br /&gt;
  pages={2541-2563},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={JMLR. org}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Visualization&lt;br /&gt;
|linkNext=Simulation }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Estimation&amp;diff=7452</id>
		<title>Estimation</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Estimation&amp;diff=7452"/>
		<updated>2013-06-25T14:20:24Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Bayesian estimation of the population parameters */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;== Introduction ==&lt;br /&gt;
&lt;br /&gt;
In the modeling context, we usually assume that we have data that includes  observations $\by$,  measurement times $\bt$ and possibly  additional regression variables $\bx$.  There may also be individual covariates $\bc$, and in pharmacological applications the dose regimen $\bu$. For clarity, in the following notation we will omit  the design variables $\bt$, $\bx$ and $\bu$, and the covariates  $\bc$.&lt;br /&gt;
&lt;br /&gt;
Here, we find ourselves in the classical framework of incomplete data models. Indeed, only $\by = (y_{ij})$ is observed in the joint model $\pypsi(\by,\bpsi;\theta)$.&lt;br /&gt;
&lt;br /&gt;
Estimation tasks are common ones seen in statistics:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; Estimate the population parameter $\theta$ using the available observations and possibly  a priori information that is available.&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Evaluate the precision of the proposed estimates.&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Reconstruct missing data, here being the individual parameters $\bpsi=(\psi_i, 1\leq i \leq N)$. &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Estimate the log-likelihood for a given model, i.e., for a given joint distribution $\qypsi$ and value of $\theta$.&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Maximum likelihood estimation of the population parameters== &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Definitions ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
''Maximum likelihood estimation''  consists of maximizing with respect to $\theta$ the ''observed likelihood''  defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\like(\theta ; \by) &amp;amp;\eqdef&amp;amp; \py(\by ; \theta) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \int \pypsi(\by,\bpsi ;\theta) \, d \bpsi .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Maximum likelihood estimation of the population parameter $\theta$ requires:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
* A model, i.e., a joint distribution $\qypsi$.  Depending on the software used, the model can be implemented using a script or a graphical user interface. $\monolix$ is extremely flexible and allows us to combine both. It is possible for instance to code the structural model using $\mlxtran$ and use the [http://en.wikipedia.org/wiki/GUI GUI] for implementing the statistical model. Whatever the options selected, the complete model can always be saved as a text file. &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* Inputs $\by$, $\bc$, $\bu$ and $\bt$.  All of these variables tend to be stored in a unique data file (see the [[Visualization#Data exploration | Data Exploration ]] Section). &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* An algorithm which allows us to maximize $\int \pypsi(\by,\bpsi ;\theta) \, d \bpsi$ with respect to $\theta$. Each software package has its own algorithms implemented. It is not our goal here to rate and compare the various algorithms and implementations. We will use exclusively the SAEM algorithm as described in [[The SAEM algorithm for estimating population parameters | The SAEM algorithm]] and implemented in $\monolix$ as we are entirely satisfied by both its theoretical and practical qualities: &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The algorithms implemented in $\monolix$ including SAEM and its extensions ([[Mixture models|mixture models]], [[Hidden Markov models|hidden Markov models]], [[Stochastic differential equations based models|SDE-based model]], [http://en.wikipedia.org/wiki/Censored_data censored data], etc.) have been published in statistical journals. Furthermore, convergence of SAEM has been rigorously proved.&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The SAEM implementation in $\monolix$ is extremely efficient for a wide variety of complex models.&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The SAEM implementation in $\monolix$ was done by the same group that proposed the algorithm and studied in detail its theoretical and practical properties.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= It is important to highlight the fact that for a parameter $\psi_i$ whose distribution is the transformation of a normal one (log-normal, logit-normal, etc.) the MLE $\hat{\psi}_{\rm pop}$ of the reference parameter $\psi_{\rm pop}$ is neither the mean nor the mode of the distribution. It is in fact the median.&lt;br /&gt;
&lt;br /&gt;
To show why this is the case, let $h$ be a nonlinear, twice continuously derivable and strictly increasing function such that $h(\psi_i)$ is normally distributed.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
*  First we show that it is not the mean. By definition, the MLE of $h(\psi_{\rm pop})$ is  $h(\hat{\psi}_{\rm pop})$. Thus, the estimated distribution of $h(\psi_i)$ is the normal distribution with mean $h(\hat{\psi}_{\rm pop})$, but $\esp{h(\psi_i)} = h(\hat{\psi}_{\rm pop})$ implies  that $\esp{\psi_i} \neq \hat{\psi}_{\rm pop}$ since $h$ is nonlinear. In other words, $\hat{\psi}_{\rm pop}$ is not the mean of the estimated distribution of $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Next we show that it is not the mode. Let $f$ be the pdf of $\psi_i$ and let $f_h$ be the pdf of $h(\psi_i)$. By definition, for any $h(t)\in \mathbb{R}$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
f(t) = h^\prime(t)f_h(h(t)) . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: Thus,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
f^\prime(t) = h^{\prime \prime}(t)f_h(h(t)) + h^{\prime 2}(t)f_h^\prime(h(t)) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: By definition of the mode, $f_h^\prime(h(\hat{\psi}_{\rm pop}))=0$. Since $h$ is nonlinear, $h^{\prime \prime}(\hat{\psi}_{\rm pop})\neq 0$ a.s. and  $f^\prime(\hat{\psi}_{\rm pop})\neq 0$ a.s.. In other words, $\hat{\psi}_{\rm pop}$ is not the mode of the estimated distribution of $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Now we show that it is the median. Since $h$ is a strictly increasing function,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\probs{\hat{\psi}_{\rm pop} }{\psi_i \leq \hat{\psi}_{\rm pop} } &amp;amp;=&amp;amp; \probs{\hat{\psi}_{\rm pop} }{h(\psi_i) \leq h(\hat{\psi}_{\rm pop})} \\&lt;br /&gt;
&amp;amp;=&amp;amp; 0.5 .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
: In other words, $\hat{\psi}_{\rm pop}$ is the median of the estimated distribution of $\psi_i$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Example ===&lt;br /&gt;
&lt;br /&gt;
Let us again look at the model used in the [[Visualization#Model exploration | Model Visualization]] Section. For the case of a unique dose $D$ given at time $t=0$, the structural model is written:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
ke&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; \displaystyle{\frac{D \, ka}{V(ka-ke)} }\left(e^{-ke\,t} - e^{-ka\,t} \right) \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $Cc$ is the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging). Supposing a constant error model for the concentration, the model for the observations can be easily implemented using $\mlxtran$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1est_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
parameter =  {ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
ke=Cl/V&lt;br /&gt;
Cc  = amtDose*ka/(V*(ka-ke))*(exp(-ke*t) - exp(-ka*t))&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&lt;br /&gt;
OBSERVATION:&lt;br /&gt;
Concentration = {type=continuous, prediction=Cc, errorModel=constant}&lt;br /&gt;
Hemorrhaging  = {type=event, hazard=h}&lt;br /&gt;
&lt;br /&gt;
OUTPUT:&lt;br /&gt;
output = {Concentration, Hemorrhaging}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, {{Verbatim|amtDose}} is a reserved keyword for the last  administered dose.&lt;br /&gt;
&lt;br /&gt;
The model's parameters are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$. The statistical model for the individual parameters can be defined in the $\monolix$ project file (left) and/or the $\monolix$ GUI (right):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INDIVIDUAL:&lt;br /&gt;
     ka    = {distribution=logNormal, iiv=yes}&lt;br /&gt;
     V     = {distribution=logNormal, iiv=yes},&lt;br /&gt;
     Cl    = {distribution=normal, iiv=yes},&lt;br /&gt;
     h0    = {distribution=probitNormal, iiv=yes},&lt;br /&gt;
     gamma = {distribution=logitNormal, iiv=yes},&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=&lt;br /&gt;
[[File:Vsaem1.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Once the model is implemented, tasks such as maximum likelihood estimation can be performed using the SAEM algorithm. Certain settings in SAEM must be provided by the user. Even though SAEM is quite insensitive to the initial parameter values,&lt;br /&gt;
it is possible to perform a preliminary sensitivity analysis in order to select &amp;quot;good&amp;quot; initial values.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=Vsaem2.png|caption=Looking for good initial values for SAEM}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Then, when we run SAEM, it converges easily and quickly to the MLE:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
              parameter&lt;br /&gt;
ka          :    0.974&lt;br /&gt;
V           :     7.07&lt;br /&gt;
Cl          :     2.00&lt;br /&gt;
h0          :   0.0102&lt;br /&gt;
gamma       :    0.485&lt;br /&gt;
&lt;br /&gt;
omega_ka    :    0.668&lt;br /&gt;
omega_V     :    0.365&lt;br /&gt;
omega_Cl    :    0.588&lt;br /&gt;
omega_h0    :    0.105&lt;br /&gt;
omega_gamma :   0.0901&lt;br /&gt;
&lt;br /&gt;
a_1         :    0.345&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Parameter estimation can therefore be seen as estimating the reference values and variance of the random effects.&lt;br /&gt;
&lt;br /&gt;
In addition to these numbers, it is important to be able to graphically represent these distributions in order to see them and therefore understand them better. In effect, the interpretation of certain parameters is not always simple. Of course, we know what a normal distribution represents and in particular its mean, median and mode, which are equal (see the distribution of $Cl$ below for instance). These measures of central tendency can be different among themselves for other asymmetric distributions such as the log-normal (see the distribution of $ka$).&lt;br /&gt;
&lt;br /&gt;
Interpreting dispersion terms like  $\omega_{ka}$  and $\omega_{V}$ is not obvious either when the parameter distributions are not normal. In such cases, quartiles or quantiles of order 5% and 95% (for example) may be useful for quantitively describing the variability of these parameters.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks &lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
For a parameter $\psi$ whose distribution is log-normal, we can approximate the coefficient of variation for $\psi$ by the standard deviation $\omega_{\psi}$ of the random effect $\eta$ if this is fairly small. In effect, when $\omega_{\psi}$ is small,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\psi &amp;amp;=&amp;amp; \psi_{\rm pop} e^{\eta} \\&lt;br /&gt;
&amp;amp;\approx &amp;amp; \psi_{\rm pop}(1+ \eta) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\esp{\psi} &amp;amp;\approx&amp;amp; \psi_{\rm pop} \\&lt;br /&gt;
\std{\psi} &amp;amp;\approx &amp;amp; \psi_{\rm pop}\omega_{\psi},&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\rm cv}(\psi) &amp;amp;=&amp;amp; \frac{\std{\psi} }{\esp{\psi} } \\&lt;br /&gt;
 &amp;amp;\approx &amp;amp; \omega_{\psi} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Do not forget that this approximation is only valid when $\omega$ is small and in the case of log-normal distributions. It does not carry over to any other distribution. Thus, when $\omega_{h0}=0.1$ for a probit-normal distribution or $\omega_{\gamma}=0.09$ for a logit-normal one, there is no immediate interpretation available. Only by looking at the graphical display of the pdf or by calculating some quantiles of interest can we begin to get an idea of dispersion in the parameters $h0$ and $\gamma$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=saem3b.png|caption=Estimation of the population distributions of the individual parameters of the model }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bayesian estimation of the population parameters==&lt;br /&gt;
&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Bayesian_probability ''Bayesian approach''] considers $\theta$ as a random vector with a ''prior distribution'' $\qth$. We can then define the posterior  distribution of $\theta$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcthy(\theta {{!}} \by ) &amp;amp;=&amp;amp; \displaystyle{ \frac{\pth( \theta )\pcyth(\by {{!}} \theta )}{\py(\by)} }\\&lt;br /&gt;
&amp;amp;=&amp;amp; \displaystyle{ \frac{\pth( \theta ) \int \pypsith(\by,\bpsi {{!}}\theta) \, d \bpsi}{\py(\by)} }.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can estimate this conditional distribution and derive any statistics (posterior mean, standard deviation, percentiles, etc.) or derive the so-called [http://en.wikipedia.org/wiki/Maximum_a_posteriori_estimation ''Maximum a Posteriori'' (MAP) estimate] of $\theta$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\theta}^{\rm MAP} &amp;amp;=&amp;amp; \argmax{\theta}  \pcthy(\theta {{!}} \by ) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmax{\theta} \left\{  {\llike}(\theta ; \by) + \log( \pth( \theta ) ) \right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The MAP estimate  therefore maximizes a penalized version of the observed likelihood.   In other words, maximum a posteriori estimation reduces to penalized maximum likelihood estimation. Suppose for instance that $\theta$ is a scalar parameter and the prior is a normal distribution with mean $\theta_0$ and variance $\gamma^2$. Then, the MAP estimate minimizes&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hat{\theta}^{\rm MAP}  =\argmax{\theta} \left\{  {\llike} (\theta ; \by) - \displaystyle{ \frac{1}{2\gamma^2} }(\theta - \theta_0)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The MAP estimate is a trade-off between the [http://en.wikipedia.org/wiki/Maximum_likelihood_estimation MLE] which maximizes ${\llike}(\theta ; \by)$ and $\theta_0$ which minimizes $(\theta - \theta_0)^2$. The weight given to the prior directly depends on the variance of the prior distribution: the smaller $\gamma^2$ is, the closer to $\theta_0$ the MAP is. The limiting distribution considers that $\gamma^2=0$: this prior means here that $\theta$ is fixed as $\theta_0$ and no longer needs to be estimated.&lt;br /&gt;
&lt;br /&gt;
Both the Bayesian and [http://en.wikipedia.org/wiki/Frequentist_probability frequentist] approaches have their supporters and detractors. But rather than being dogmatic and blindly following the same rule-book every time, we need to be pragmatic and ask the right methodological questions when confronted with a new problem.&lt;br /&gt;
&lt;br /&gt;
We have to remember that Bayesian methods have been extremely successful, in particular for numerical calculations. For instance, (Bayesian) [http://en.wikipedia.org/wiki/Markov_chain_Monte_Carlo MCMC methods] allow us to estimate more or less any conditional distribution coming from any hierarchical model, whereas frequentist approaches such as maximum likelihood estimation can be much more difficult to implement.&lt;br /&gt;
&lt;br /&gt;
All things said, the problem comes down to knowing whether the data contains sufficient information to answer a given question, and whether some other information may be available to help answer it. This is the essence of the art of modeling: finding the right compromise between the confidence we have in the data and prior knowledge of the problem. Each problem is different and requires a specific approach. For instance, if all the patients in a pharmacokinetic trial have essentially the same weight, it is pointless to estimate a relationship between weight and the model's PK parameters using the trial data. In this case, the modeler would be better served trying to use prior information based on physiological criteria rather than just a statistical model.&lt;br /&gt;
&lt;br /&gt;
Therefore, we can use information available to us, of course! Why not? But this information needs to be pertinent. Systematically using a prior for the parameters is not always meaningful. Can we reasonable suppose that we have access to such information? For continuous data for example, what does putting a prior on the residual error model's parameters mean in reality? A reasoned statistical approach consists of only including prior information for certain parameters (those for which we have real prior information) and having confidence in the data for the others.&lt;br /&gt;
&lt;br /&gt;
$\monolix$ allows this hybrid approach which  reconciles the Bayesian and frequentist approaches. A given parameter can be:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* a fixed constant if we have absolute confidence in its value or the data does not allow it to be estimated, essentially due to identifiability constraints.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by maximum likelihood, either because we have great confidence in the data or have no information on the parameter.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by introducing a prior and calculating the MAP estimate.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by introducing a prior and then estimating the posterior distribution.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We put aside dealing with the fixed components of $\theta$ in the following. Here are some possible situations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; ''Combined maximum likelihood and maximum a posteriori estimation'':  decompose $\theta$ into $(\theta_E,\theta_{M})$ where $\theta_E$ are the components of $\theta$ to be estimated with MLE and $\theta_{M}$ those with a prior distribution whose posterior distribution is to be maximized. Then,  $(\hat{\theta}_E , \hat{\theta}_{M} )$ below  maximizes the penalized  likelihood of $(\theta_E,\theta_{M})$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
(\hat{\theta}_E , \hat{\theta}_{M} )  &amp;amp;=&amp;amp; \argmax{\theta_E , \theta_{M} } \log(\py(\by , \theta_{M}; \theta_E)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \argmax{\theta_E , \theta_{M} }  \left\{  {\llike}(\theta_E , \theta_{M}; \by) + \log( \pth( \theta_M ) ) \right\} ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where ${\llike} (\theta_E , \theta_{M}; \by) \ \  \eqdef \ \ \log\left(\py(\by | \theta_{M}; \theta_E)\right).$&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; ''Combined maximum likelihood and posterior distribution estimation'': here, decompose  $\theta$ into $(\theta_E,\theta_{R})$ where $\theta_E$ are the components of $\theta$ to be estimated with MLE and $\theta_{R}$ those with a prior distribution whose posterior distribution is to be estimated. We  propose the following strategy for estimating $\theta_E$ and $\theta_{R}$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol style=&amp;quot;list-style-type:lower-roman&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; Compute the maximum likelihood of $\theta_E$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\theta}_E  &amp;amp;=&amp;amp; \argmax{\theta_E}  \log(\py(\by ; \theta_E)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \argmax{\theta_E}  \int \pmacro(\by , \theta_R ; \theta_E ) d \theta_R .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; Estimate the conditional distribution $\pmacro(\theta_{R} | \by ;\hat{\theta}_E)$. &amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is then straightforward to extend this approach to more complex situations where  some components of $\theta$ are estimated with MLE, others using MAP estimation and others still by estimating their conditional distributions.&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A PK example&lt;br /&gt;
|text=&lt;br /&gt;
In this example we use only the pharmacokinetic data and aim to estimate the population parameter distributions of the PK parameters $ka$, $V$ and $Cl$. We assume log-normal distributions for these three parameters. All of the model's population parameters are estimated by maximum likelihood estimation except $ka_{\rm pop}$ for which a log-normal distribution is used as a prior:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \log(ka_{\rm pop}) \sim {\cal N}(\log(1.5), \gamma^2) . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
$\monolix$ allows us to compute the MAP estimate and to estimate the posterior distribution of $ka_{\rm pop}$ for various values of $\gamma$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;margin-left:15%; margin-right:32%; align:center&amp;quot;&amp;gt;&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot;  align=&amp;quot;center&amp;quot; style=&amp;quot;width:100%&amp;quot;&lt;br /&gt;
{{!}} $\gamma$ {{!}}{{!}} 0 {{!}}{{!}} 0.01 {{!}}{{!}} 0.025 {{!}}{{!}} 0.05 {{!}}{{!}} 0.1 {{!}}{{!}} 0.2 {{!}}{{!}} $+ \infty$ &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}$\hat{ka}_{\rm pop}^{\rm MAP}$ {{!}}{{!}} 1.5 {{!}}{{!}} 1.49 {{!}}{{!}} 1.47 {{!}}{{!}} 1.39 {{!}}{{!}} 1.22 {{!}}{{!}} 1.11 {{!}}{{!}} 1.05 &lt;br /&gt;
{{!}}}&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=bayes1.png|caption=Prior and posterior distributions of $ka_{\rm pop}$ for different values of $\gamma$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As expected, the posterior distribution converges to the prior distribution when the  standard deviation $\gamma$ of the prior distribution decreases. Also, the mode of the posterior distribution converges to the maximum likelihood estimate of $ka_{\rm pop}$ when $\gamma$ increases.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation of the Fisher information matrix ==&lt;br /&gt;
&lt;br /&gt;
The variance of the estimator $\thmle$ and thus confidence intervals can be derived  from the [[Estimation of the observed Fisher information matrix|observed Fisher information matrix (F.I.M.)]], which itself is calculated using the observed likelihood (i.e., the pdf of the observations $\by$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;ofim_intro3&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\ofim(\thmle ; \by) \ \ \eqdef \ \ - \displaystyle{ \frac{\partial^2}{\partial \theta^2} }\log({\like}(\thmle ; \by)) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1)  }}&lt;br /&gt;
&lt;br /&gt;
Then, the variance-covariance matrix of the maximum likelihood estimator $\thmle$ can be estimated by the inverse of the observed F.I.M. Standard errors (s.e.) for each component of $\thmle$ are their standard deviations, i.e., the square-root of the diagonal elements of this covariance matrix. $\monolix$ also displays the (estimated) relative standard errors (r.s.e.), i.e., the (estimated) standard error divided by the value of the estimated parameter.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
              parameter     s.e. (s.a.)   r.s.e.(%)&lt;br /&gt;
ka          :    0.974         0.082           8&lt;br /&gt;
V           :     7.07          0.35           5&lt;br /&gt;
Cl          :        2          0.07           4&lt;br /&gt;
h0          :   0.0102        0.0014          14&lt;br /&gt;
gamma       :    0.485         0.015           3&lt;br /&gt;
&lt;br /&gt;
omega_ka    :    0.668         0.064          10&lt;br /&gt;
omega_V     :    0.365         0.037          10&lt;br /&gt;
omega_Cl    :    0.588         0.055           9&lt;br /&gt;
omega_h0    :    0.105         0.032          30&lt;br /&gt;
omega_gamma :   0.0901         0.044          49&lt;br /&gt;
&lt;br /&gt;
a_1         :    0.345         0.012           3&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The F.I.M. can be used for detecting overparametrization of the structural model. In effect, if the model is poorly identifiable,  certain estimators will be quite correlated and the F.I.M. will therefore be poorly conditioned and difficult to inverse. Suppose for example that we want to fit a two compartment PK model to the same data as before. The output is shown below. The large values for the relative standard errors for the inter-compartmental clearance $Q$ and the volume of the peripheral compartment $V_2$ mean that the data does not allow us to  estimate well these two parameters.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
           parameter     s.e. (lin)   r.s.e.(%)&lt;br /&gt;
ka       :     0.246       0.0081             3&lt;br /&gt;
Cl       :       1.9        0.075             4&lt;br /&gt;
V1       :      1.71         0.14             8&lt;br /&gt;
Q        :  0.000171        0.024      1.43e+04&lt;br /&gt;
V2       :   0.00673          3.1      4.62e+04&lt;br /&gt;
&lt;br /&gt;
omega_ka :     0.171        0.026            15&lt;br /&gt;
omega_Cl :     0.293        0.026             9&lt;br /&gt;
omega_V1 :     0.621        0.062            10&lt;br /&gt;
omega_Q  :      5.72      1.4e+03      2.41e+04&lt;br /&gt;
omega_V2 :      4.61      1.8e+04      3.94e+05&lt;br /&gt;
&lt;br /&gt;
a        :     0.136       0.0073             5&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The Fisher information criteria is also widely used in optimal experimental design. Indeed, minimizing the variance of the estimator corresponds to maximizing the information.  Then, estimators and designs can be evaluated by looking at certain summary statistics of the covariance matrix (like the determinant or trace for instance).&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Estimation of the individual parameters ==&lt;br /&gt;
&lt;br /&gt;
Once $\theta$ has been estimated, the conditional distribution $\pmacro(\psi_i | y_i ; \hat{\theta})$ of the individual parameters $\psi_i$ can be estimated for each individual $i$ using the [[The Metropolis-Hastings algorithm for simulating the individual parameters| Metropolis-Hastings algorithm]]. For each $i$, this algorithm generates a sequence $(\psi_i^{k}, k \geq 1)$ which converges in distribution to the conditional distribution $\pmacro(\psi_i | y_i ; \hat{\theta})$  and that can be used for estimating any summary statistic of this distribution (mean, standard deviation, quantiles, etc.).&lt;br /&gt;
&lt;br /&gt;
The mode of this conditional distribution can be estimated using this sequence or by maximizing $\pmacro(\psi_i | y_i ; \hat{\theta})$ using numerical methods.&lt;br /&gt;
&lt;br /&gt;
The choice of using the conditional mean or the conditional mode is arbitrary. By default, $\monolix$ uses the conditional mode, taking the philosophy that the &amp;quot;most likely&amp;quot;  values of the individual parameters are the most suited for computing the &amp;quot;most likely&amp;quot; predictions.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=mode1.png|caption=Predicted concentrations for 6 individuals using the estimated  conditional modes of the individual PK parameters}} &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation of the observed log-likelihood ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Once $\theta$ has been estimated, the observed log-likelihood of $\hat{\theta}$ is defined as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
{\llike} (\hat{\theta};\by) &amp;amp;=&amp;amp; \log({\like}(\hat{\theta};\by)) \\&lt;br /&gt;
&amp;amp;\eqdef&amp;amp; \log(\py(\by;\hat{\theta})) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The observed log-likelihood cannot be computed in closed form for nonlinear mixed effects models, but can be estimated using the methods described in the [[Estimation of the log-likelihood]] Section. The estimated  log-likelihood can then be used for performing likelihood ratio tests and for computing information criteria such as AIC and BIC (see the [[Model evaluation]] Section).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Monolix,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {Monolix 4.2},&lt;br /&gt;
year={2012}&lt;br /&gt;
journal = {http://www.lixoft.eu/products/monolix/product-monolix-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{comets2011package,&lt;br /&gt;
  title={saemix: Stochastic Approximation Expectation Maximization (SAEM) algorithm. R package version 0.96.1},&lt;br /&gt;
  author={Comets, E. and Lavenu, A. and Lavielle, M.},&lt;br /&gt;
journal = {http://cran.r-project.org/web/packages/saemix/index.html},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{nlmefitsa,&lt;br /&gt;
  title={nlmefitsa: fit nonlinear mixed-effects model with stochastic EM algorithm. Matlab R2013a function},&lt;br /&gt;
  author={The MathWorks},&lt;br /&gt;
journal = {http://www.mathworks.fr/fr/help/stats/nlmefitsa.html},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{beal1992nonmem,&lt;br /&gt;
  title={NONMEM users guides},&lt;br /&gt;
  author={Beal, S.L. and Sheiner, L.B. and Boeckmann, A. and Bauer, R.J.},&lt;br /&gt;
  journal={San Francisco, NONMEM Project Group, University of California},&lt;br /&gt;
  year={1992}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{pinheiro2000mixed,&lt;br /&gt;
  title={Mixed effects models in S and S-PLUS},&lt;br /&gt;
  author={Pinheiro, J.C. and Bates, D.M.},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Springer Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{pinheiro2010r,&lt;br /&gt;
  title={the R Core team (2009) nlme: Linear and Nonlinear Mixed Effects Models. R package version 3.1-96},&lt;br /&gt;
  author={Pinheiro, J. and Bates, D. and DebRoy, S. and Sarkar, D.},&lt;br /&gt;
  journal={R Foundation for Statistical Computing, Vienna},&lt;br /&gt;
  year={2010}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{spiegelhalter2003winbugs,&lt;br /&gt;
  title={WinBUGS user manual},&lt;br /&gt;
  author={Spiegelhalter, D. and Thomas, A. and Best, N. and Lunn, D.},&lt;br /&gt;
  journal={Cambridge: MRC Biostatistics Unit},&lt;br /&gt;
  year={2003}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Manual{docSPSS,&lt;br /&gt;
		title = {Linear mixed-effects modeling in SPSS. An introduction to the MIXED procedure},&lt;br /&gt;
    author = {SPSS},&lt;br /&gt;
    year = {2002},&lt;br /&gt;
    note={Technical Report}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Manual{docSAS,&lt;br /&gt;
		title = {The NLMMIXED procedure, SAS/STAT 9.2 User's Guide},&lt;br /&gt;
		chapter = {61},&lt;br /&gt;
		pages = {4337--4435},&lt;br /&gt;
		author = {SAS},&lt;br /&gt;
		year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Visualization&lt;br /&gt;
|linkNext=Model evaluation }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Estimation&amp;diff=7451</id>
		<title>Estimation</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Estimation&amp;diff=7451"/>
		<updated>2013-06-25T14:15:01Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Definitions */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;== Introduction ==&lt;br /&gt;
&lt;br /&gt;
In the modeling context, we usually assume that we have data that includes  observations $\by$,  measurement times $\bt$ and possibly  additional regression variables $\bx$.  There may also be individual covariates $\bc$, and in pharmacological applications the dose regimen $\bu$. For clarity, in the following notation we will omit  the design variables $\bt$, $\bx$ and $\bu$, and the covariates  $\bc$.&lt;br /&gt;
&lt;br /&gt;
Here, we find ourselves in the classical framework of incomplete data models. Indeed, only $\by = (y_{ij})$ is observed in the joint model $\pypsi(\by,\bpsi;\theta)$.&lt;br /&gt;
&lt;br /&gt;
Estimation tasks are common ones seen in statistics:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; Estimate the population parameter $\theta$ using the available observations and possibly  a priori information that is available.&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Evaluate the precision of the proposed estimates.&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Reconstruct missing data, here being the individual parameters $\bpsi=(\psi_i, 1\leq i \leq N)$. &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Estimate the log-likelihood for a given model, i.e., for a given joint distribution $\qypsi$ and value of $\theta$.&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Maximum likelihood estimation of the population parameters== &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Definitions ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
''Maximum likelihood estimation''  consists of maximizing with respect to $\theta$ the ''observed likelihood''  defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\like(\theta ; \by) &amp;amp;\eqdef&amp;amp; \py(\by ; \theta) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \int \pypsi(\by,\bpsi ;\theta) \, d \bpsi .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Maximum likelihood estimation of the population parameter $\theta$ requires:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
* A model, i.e., a joint distribution $\qypsi$.  Depending on the software used, the model can be implemented using a script or a graphical user interface. $\monolix$ is extremely flexible and allows us to combine both. It is possible for instance to code the structural model using $\mlxtran$ and use the [http://en.wikipedia.org/wiki/GUI GUI] for implementing the statistical model. Whatever the options selected, the complete model can always be saved as a text file. &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* Inputs $\by$, $\bc$, $\bu$ and $\bt$.  All of these variables tend to be stored in a unique data file (see the [[Visualization#Data exploration | Data Exploration ]] Section). &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* An algorithm which allows us to maximize $\int \pypsi(\by,\bpsi ;\theta) \, d \bpsi$ with respect to $\theta$. Each software package has its own algorithms implemented. It is not our goal here to rate and compare the various algorithms and implementations. We will use exclusively the SAEM algorithm as described in [[The SAEM algorithm for estimating population parameters | The SAEM algorithm]] and implemented in $\monolix$ as we are entirely satisfied by both its theoretical and practical qualities: &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The algorithms implemented in $\monolix$ including SAEM and its extensions ([[Mixture models|mixture models]], [[Hidden Markov models|hidden Markov models]], [[Stochastic differential equations based models|SDE-based model]], [http://en.wikipedia.org/wiki/Censored_data censored data], etc.) have been published in statistical journals. Furthermore, convergence of SAEM has been rigorously proved.&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The SAEM implementation in $\monolix$ is extremely efficient for a wide variety of complex models.&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The SAEM implementation in $\monolix$ was done by the same group that proposed the algorithm and studied in detail its theoretical and practical properties.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= It is important to highlight the fact that for a parameter $\psi_i$ whose distribution is the transformation of a normal one (log-normal, logit-normal, etc.) the MLE $\hat{\psi}_{\rm pop}$ of the reference parameter $\psi_{\rm pop}$ is neither the mean nor the mode of the distribution. It is in fact the median.&lt;br /&gt;
&lt;br /&gt;
To show why this is the case, let $h$ be a nonlinear, twice continuously derivable and strictly increasing function such that $h(\psi_i)$ is normally distributed.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
*  First we show that it is not the mean. By definition, the MLE of $h(\psi_{\rm pop})$ is  $h(\hat{\psi}_{\rm pop})$. Thus, the estimated distribution of $h(\psi_i)$ is the normal distribution with mean $h(\hat{\psi}_{\rm pop})$, but $\esp{h(\psi_i)} = h(\hat{\psi}_{\rm pop})$ implies  that $\esp{\psi_i} \neq \hat{\psi}_{\rm pop}$ since $h$ is nonlinear. In other words, $\hat{\psi}_{\rm pop}$ is not the mean of the estimated distribution of $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Next we show that it is not the mode. Let $f$ be the pdf of $\psi_i$ and let $f_h$ be the pdf of $h(\psi_i)$. By definition, for any $h(t)\in \mathbb{R}$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
f(t) = h^\prime(t)f_h(h(t)) . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: Thus,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
f^\prime(t) = h^{\prime \prime}(t)f_h(h(t)) + h^{\prime 2}(t)f_h^\prime(h(t)) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: By definition of the mode, $f_h^\prime(h(\hat{\psi}_{\rm pop}))=0$. Since $h$ is nonlinear, $h^{\prime \prime}(\hat{\psi}_{\rm pop})\neq 0$ a.s. and  $f^\prime(\hat{\psi}_{\rm pop})\neq 0$ a.s.. In other words, $\hat{\psi}_{\rm pop}$ is not the mode of the estimated distribution of $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Now we show that it is the median. Since $h$ is a strictly increasing function,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\probs{\hat{\psi}_{\rm pop} }{\psi_i \leq \hat{\psi}_{\rm pop} } &amp;amp;=&amp;amp; \probs{\hat{\psi}_{\rm pop} }{h(\psi_i) \leq h(\hat{\psi}_{\rm pop})} \\&lt;br /&gt;
&amp;amp;=&amp;amp; 0.5 .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
: In other words, $\hat{\psi}_{\rm pop}$ is the median of the estimated distribution of $\psi_i$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Example ===&lt;br /&gt;
&lt;br /&gt;
Let us again look at the model used in the [[Visualization#Model exploration | Model Visualization]] Section. For the case of a unique dose $D$ given at time $t=0$, the structural model is written:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
ke&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; \displaystyle{\frac{D \, ka}{V(ka-ke)} }\left(e^{-ke\,t} - e^{-ka\,t} \right) \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $Cc$ is the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging). Supposing a constant error model for the concentration, the model for the observations can be easily implemented using $\mlxtran$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1est_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
parameter =  {ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
ke=Cl/V&lt;br /&gt;
Cc  = amtDose*ka/(V*(ka-ke))*(exp(-ke*t) - exp(-ka*t))&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&lt;br /&gt;
OBSERVATION:&lt;br /&gt;
Concentration = {type=continuous, prediction=Cc, errorModel=constant}&lt;br /&gt;
Hemorrhaging  = {type=event, hazard=h}&lt;br /&gt;
&lt;br /&gt;
OUTPUT:&lt;br /&gt;
output = {Concentration, Hemorrhaging}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, {{Verbatim|amtDose}} is a reserved keyword for the last  administered dose.&lt;br /&gt;
&lt;br /&gt;
The model's parameters are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$. The statistical model for the individual parameters can be defined in the $\monolix$ project file (left) and/or the $\monolix$ GUI (right):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INDIVIDUAL:&lt;br /&gt;
     ka    = {distribution=logNormal, iiv=yes}&lt;br /&gt;
     V     = {distribution=logNormal, iiv=yes},&lt;br /&gt;
     Cl    = {distribution=normal, iiv=yes},&lt;br /&gt;
     h0    = {distribution=probitNormal, iiv=yes},&lt;br /&gt;
     gamma = {distribution=logitNormal, iiv=yes},&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=&lt;br /&gt;
[[File:Vsaem1.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Once the model is implemented, tasks such as maximum likelihood estimation can be performed using the SAEM algorithm. Certain settings in SAEM must be provided by the user. Even though SAEM is quite insensitive to the initial parameter values,&lt;br /&gt;
it is possible to perform a preliminary sensitivity analysis in order to select &amp;quot;good&amp;quot; initial values.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=Vsaem2.png|caption=Looking for good initial values for SAEM}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Then, when we run SAEM, it converges easily and quickly to the MLE:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
              parameter&lt;br /&gt;
ka          :    0.974&lt;br /&gt;
V           :     7.07&lt;br /&gt;
Cl          :     2.00&lt;br /&gt;
h0          :   0.0102&lt;br /&gt;
gamma       :    0.485&lt;br /&gt;
&lt;br /&gt;
omega_ka    :    0.668&lt;br /&gt;
omega_V     :    0.365&lt;br /&gt;
omega_Cl    :    0.588&lt;br /&gt;
omega_h0    :    0.105&lt;br /&gt;
omega_gamma :   0.0901&lt;br /&gt;
&lt;br /&gt;
a_1         :    0.345&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Parameter estimation can therefore be seen as estimating the reference values and variance of the random effects.&lt;br /&gt;
&lt;br /&gt;
In addition to these numbers, it is important to be able to graphically represent these distributions in order to see them and therefore understand them better. In effect, the interpretation of certain parameters is not always simple. Of course, we know what a normal distribution represents and in particular its mean, median and mode, which are equal (see the distribution of $Cl$ below for instance). These measures of central tendency can be different among themselves for other asymmetric distributions such as the log-normal (see the distribution of $ka$).&lt;br /&gt;
&lt;br /&gt;
Interpreting dispersion terms like  $\omega_{ka}$  and $\omega_{V}$ is not obvious either when the parameter distributions are not normal. In such cases, quartiles or quantiles of order 5% and 95% (for example) may be useful for quantitively describing the variability of these parameters.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks &lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
For a parameter $\psi$ whose distribution is log-normal, we can approximate the coefficient of variation for $\psi$ by the standard deviation $\omega_{\psi}$ of the random effect $\eta$ if this is fairly small. In effect, when $\omega_{\psi}$ is small,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\psi &amp;amp;=&amp;amp; \psi_{\rm pop} e^{\eta} \\&lt;br /&gt;
&amp;amp;\approx &amp;amp; \psi_{\rm pop}(1+ \eta) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\esp{\psi} &amp;amp;\approx&amp;amp; \psi_{\rm pop} \\&lt;br /&gt;
\std{\psi} &amp;amp;\approx &amp;amp; \psi_{\rm pop}\omega_{\psi},&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\rm cv}(\psi) &amp;amp;=&amp;amp; \frac{\std{\psi} }{\esp{\psi} } \\&lt;br /&gt;
 &amp;amp;\approx &amp;amp; \omega_{\psi} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Do not forget that this approximation is only valid when $\omega$ is small and in the case of log-normal distributions. It does not carry over to any other distribution. Thus, when $\omega_{h0}=0.1$ for a probit-normal distribution or $\omega_{\gamma}=0.09$ for a logit-normal one, there is no immediate interpretation available. Only by looking at the graphical display of the pdf or by calculating some quantiles of interest can we begin to get an idea of dispersion in the parameters $h0$ and $\gamma$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=saem3b.png|caption=Estimation of the population distributions of the individual parameters of the model }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bayesian estimation of the population parameters==&lt;br /&gt;
&lt;br /&gt;
The ''Bayesian approach'' considers $\theta$ as a random vector with a ''prior distribution'' $\qth$. We can then define the posterior  distribution of $\theta$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcthy(\theta {{!}} \by ) &amp;amp;=&amp;amp; \displaystyle{ \frac{\pth( \theta )\pcyth(\by {{!}} \theta )}{\py(\by)} }\\&lt;br /&gt;
&amp;amp;=&amp;amp; \displaystyle{ \frac{\pth( \theta ) \int \pypsith(\by,\bpsi {{!}}\theta) \, d \bpsi}{\py(\by)} }.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can estimate this conditional distribution and derive any statistics (posterior mean, standard deviation, percentiles, etc.) or derive the so-called ''Maximum a Posteriori'' (MAP) estimate of $\theta$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\theta}^{\rm MAP} &amp;amp;=&amp;amp; \argmax{\theta}  \pcthy(\theta {{!}} \by ) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmax{\theta} \left\{  {\llike}(\theta ; \by) + \log( \pth( \theta ) ) \right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The MAP estimate  therefore maximizes a penalized version of the observed likelihood.   In other words, maximum a posteriori estimation reduces to penalized maximum likelihood estimation. Suppose for instance that $\theta$ is a scalar parameter and the prior is a normal distribution with mean $\theta_0$ and variance $\gamma^2$. Then, the MAP estimate minimizes&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hat{\theta}^{\rm MAP}  =\argmax{\theta} \left\{  {\llike} (\theta ; \by) - \displaystyle{ \frac{1}{2\gamma^2} }(\theta - \theta_0)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The MAP estimate is a trade-off between the MLE which maximizes ${\llike}(\theta ; \by)$ and $\theta_0$ which minimizes $(\theta - \theta_0)^2$. The weight given to the prior directly depends on the variance of the prior distribution: the smaller $\gamma^2$ is, the closer to $\theta_0$ the MAP is. The limiting distribution considers that $\gamma^2=0$: this prior means here that $\theta$ is fixed as $\theta_0$ and no longer needs to be estimated.&lt;br /&gt;
&lt;br /&gt;
Both the Bayesian and frequentist approaches have their supporters and detractors. But rather than being dogmatic and blindly following the same rule-book every time, we need to be pragmatic and ask the right methodological questions when confronted with a new problem.&lt;br /&gt;
&lt;br /&gt;
We have to remember that Bayesian methods have been extremely successful, in particular for numerical calculations. For instance, (Bayesian) MCMC methods allow us to estimate more or less any conditional distribution coming from any hierarchical model, whereas frequentist approaches such as maximum likelihood estimation can be much more difficult to implement.&lt;br /&gt;
&lt;br /&gt;
All things said, the problem comes down to knowing whether the data contains sufficient information to answer a given question, and whether some other information may be available to help answer it. This is the essence of the art of modeling: finding the right compromise between the confidence we have in the data and prior knowledge of the problem. Each problem is different and requires a specific approach. For instance, if all the patients in a pharmacokinetic trial have essentially the same weight, it is pointless to estimate a relationship between weight and the model's PK parameters using the trial data. In this case, the modeler would be better served trying to use prior information based on physiological criteria rather than just a statistical model.&lt;br /&gt;
&lt;br /&gt;
Therefore, we can use information available to us, of course! Why not? But this information needs to be pertinent. Systematically using a prior for the parameters is not always meaningful. Can we reasonable suppose that we have access to such information? For continuous data for example, what does putting a prior on the residual error model's parameters mean in reality? A reasoned statistical approach consists of only including prior information for certain parameters (those for which we have real prior information) and having confidence in the data for the others.&lt;br /&gt;
&lt;br /&gt;
$\monolix$ allows this hybrid approach which  reconciles the Bayesian and frequentist approaches. A given parameter can be:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* a fixed constant if we have absolute confidence in its value or the data does not allow it to be estimated, essentially due to identifiability constraints.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by maximum likelihood, either because we have great confidence in the data or have no information on the parameter.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by introducing a prior and calculating the MAP estimate.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by introducing a prior and then estimating the posterior distribution.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We put aside dealing with the fixed components of $\theta$ in the following. Here are some possible situations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; ''Combined maximum likelihood and maximum a posteriori estimation'':  decompose $\theta$ into $(\theta_E,\theta_{M})$ where $\theta_E$ are the components of $\theta$ to be estimated with MLE and $\theta_{M}$ those with a prior distribution whose posterior distribution is to be maximized. Then,  $(\hat{\theta}_E , \hat{\theta}_{M} )$ below  maximizes the penalized  likelihood of $(\theta_E,\theta_{M})$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
(\hat{\theta}_E , \hat{\theta}_{M} )  &amp;amp;=&amp;amp; \argmax{\theta_E , \theta_{M} } \log(\py(\by , \theta_{M}; \theta_E)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \argmax{\theta_E , \theta_{M} }  \left\{  {\llike}(\theta_E , \theta_{M}; \by) + \log( \pth( \theta_M ) ) \right\} ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where ${\llike} (\theta_E , \theta_{M}; \by) \ \  \eqdef \ \ \log\left(\py(\by | \theta_{M}; \theta_E)\right).$&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; ''Combined maximum likelihood and posterior distribution estimation'': here, decompose  $\theta$ into $(\theta_E,\theta_{R})$ where $\theta_E$ are the components of $\theta$ to be estimated with MLE and $\theta_{R}$ those with a prior distribution whose posterior distribution is to be estimated. We  propose the following strategy for estimating $\theta_E$ and $\theta_{R}$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol style=&amp;quot;list-style-type:lower-roman&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; Compute the maximum likelihood of $\theta_E$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\theta}_E  &amp;amp;=&amp;amp; \argmax{\theta_E}  \log(\py(\by ; \theta_E)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \argmax{\theta_E}  \int \pmacro(\by , \theta_R ; \theta_E ) d \theta_R .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; Estimate the conditional distribution $\pmacro(\theta_{R} | \by ;\hat{\theta}_E)$. &amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is then straightforward to extend this approach to more complex situations where  some components of $\theta$ are estimated with MLE, others using MAP estimation and others still by estimating their conditional distributions.&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A PK example&lt;br /&gt;
|text=&lt;br /&gt;
In this example we use only the pharmacokinetic data and aim to estimate the population parameter distributions of the PK parameters $ka$, $V$ and $Cl$. We assume log-normal distributions for these three parameters. All of the model's population parameters are estimated by maximum likelihood estimation except $ka_{\rm pop}$ for which a log-normal distribution is used as a prior:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \log(ka_{\rm pop}) \sim {\cal N}(\log(1.5), \gamma^2) . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
$\monolix$ allows us to compute the MAP estimate and to estimate the posterior distribution of $ka_{\rm pop}$ for various values of $\gamma$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;margin-left:15%; margin-right:32%; align:center&amp;quot;&amp;gt;&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot;  align=&amp;quot;center&amp;quot; style=&amp;quot;width:100%&amp;quot;&lt;br /&gt;
{{!}} $\gamma$ {{!}}{{!}} 0 {{!}}{{!}} 0.01 {{!}}{{!}} 0.025 {{!}}{{!}} 0.05 {{!}}{{!}} 0.1 {{!}}{{!}} 0.2 {{!}}{{!}} $+ \infty$ &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}$\hat{ka}_{\rm pop}^{\rm MAP}$ {{!}}{{!}} 1.5 {{!}}{{!}} 1.49 {{!}}{{!}} 1.47 {{!}}{{!}} 1.39 {{!}}{{!}} 1.22 {{!}}{{!}} 1.11 {{!}}{{!}} 1.05 &lt;br /&gt;
{{!}}}&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=bayes1.png|caption=Prior and posterior distributions of $ka_{\rm pop}$ for different values of $\gamma$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As expected, the posterior distribution converges to the prior distribution when the  standard deviation $\gamma$ of the prior distribution decreases. Also, the mode of the posterior distribution converges to the maximum likelihood estimate of $ka_{\rm pop}$ when $\gamma$ increases.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Estimation of the Fisher information matrix ==&lt;br /&gt;
&lt;br /&gt;
The variance of the estimator $\thmle$ and thus confidence intervals can be derived  from the [[Estimation of the observed Fisher information matrix|observed Fisher information matrix (F.I.M.)]], which itself is calculated using the observed likelihood (i.e., the pdf of the observations $\by$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;ofim_intro3&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\ofim(\thmle ; \by) \ \ \eqdef \ \ - \displaystyle{ \frac{\partial^2}{\partial \theta^2} }\log({\like}(\thmle ; \by)) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1)  }}&lt;br /&gt;
&lt;br /&gt;
Then, the variance-covariance matrix of the maximum likelihood estimator $\thmle$ can be estimated by the inverse of the observed F.I.M. Standard errors (s.e.) for each component of $\thmle$ are their standard deviations, i.e., the square-root of the diagonal elements of this covariance matrix. $\monolix$ also displays the (estimated) relative standard errors (r.s.e.), i.e., the (estimated) standard error divided by the value of the estimated parameter.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
              parameter     s.e. (s.a.)   r.s.e.(%)&lt;br /&gt;
ka          :    0.974         0.082           8&lt;br /&gt;
V           :     7.07          0.35           5&lt;br /&gt;
Cl          :        2          0.07           4&lt;br /&gt;
h0          :   0.0102        0.0014          14&lt;br /&gt;
gamma       :    0.485         0.015           3&lt;br /&gt;
&lt;br /&gt;
omega_ka    :    0.668         0.064          10&lt;br /&gt;
omega_V     :    0.365         0.037          10&lt;br /&gt;
omega_Cl    :    0.588         0.055           9&lt;br /&gt;
omega_h0    :    0.105         0.032          30&lt;br /&gt;
omega_gamma :   0.0901         0.044          49&lt;br /&gt;
&lt;br /&gt;
a_1         :    0.345         0.012           3&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The F.I.M. can be used for detecting overparametrization of the structural model. In effect, if the model is poorly identifiable,  certain estimators will be quite correlated and the F.I.M. will therefore be poorly conditioned and difficult to inverse. Suppose for example that we want to fit a two compartment PK model to the same data as before. The output is shown below. The large values for the relative standard errors for the inter-compartmental clearance $Q$ and the volume of the peripheral compartment $V_2$ mean that the data does not allow us to  estimate well these two parameters.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
           parameter     s.e. (lin)   r.s.e.(%)&lt;br /&gt;
ka       :     0.246       0.0081             3&lt;br /&gt;
Cl       :       1.9        0.075             4&lt;br /&gt;
V1       :      1.71         0.14             8&lt;br /&gt;
Q        :  0.000171        0.024      1.43e+04&lt;br /&gt;
V2       :   0.00673          3.1      4.62e+04&lt;br /&gt;
&lt;br /&gt;
omega_ka :     0.171        0.026            15&lt;br /&gt;
omega_Cl :     0.293        0.026             9&lt;br /&gt;
omega_V1 :     0.621        0.062            10&lt;br /&gt;
omega_Q  :      5.72      1.4e+03      2.41e+04&lt;br /&gt;
omega_V2 :      4.61      1.8e+04      3.94e+05&lt;br /&gt;
&lt;br /&gt;
a        :     0.136       0.0073             5&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The Fisher information criteria is also widely used in optimal experimental design. Indeed, minimizing the variance of the estimator corresponds to maximizing the information.  Then, estimators and designs can be evaluated by looking at certain summary statistics of the covariance matrix (like the determinant or trace for instance).&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Estimation of the individual parameters ==&lt;br /&gt;
&lt;br /&gt;
Once $\theta$ has been estimated, the conditional distribution $\pmacro(\psi_i | y_i ; \hat{\theta})$ of the individual parameters $\psi_i$ can be estimated for each individual $i$ using the [[The Metropolis-Hastings algorithm for simulating the individual parameters| Metropolis-Hastings algorithm]]. For each $i$, this algorithm generates a sequence $(\psi_i^{k}, k \geq 1)$ which converges in distribution to the conditional distribution $\pmacro(\psi_i | y_i ; \hat{\theta})$  and that can be used for estimating any summary statistic of this distribution (mean, standard deviation, quantiles, etc.).&lt;br /&gt;
&lt;br /&gt;
The mode of this conditional distribution can be estimated using this sequence or by maximizing $\pmacro(\psi_i | y_i ; \hat{\theta})$ using numerical methods.&lt;br /&gt;
&lt;br /&gt;
The choice of using the conditional mean or the conditional mode is arbitrary. By default, $\monolix$ uses the conditional mode, taking the philosophy that the &amp;quot;most likely&amp;quot;  values of the individual parameters are the most suited for computing the &amp;quot;most likely&amp;quot; predictions.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=mode1.png|caption=Predicted concentrations for 6 individuals using the estimated  conditional modes of the individual PK parameters}} &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation of the observed log-likelihood ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Once $\theta$ has been estimated, the observed log-likelihood of $\hat{\theta}$ is defined as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
{\llike} (\hat{\theta};\by) &amp;amp;=&amp;amp; \log({\like}(\hat{\theta};\by)) \\&lt;br /&gt;
&amp;amp;\eqdef&amp;amp; \log(\py(\by;\hat{\theta})) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The observed log-likelihood cannot be computed in closed form for nonlinear mixed effects models, but can be estimated using the methods described in the [[Estimation of the log-likelihood]] Section. The estimated  log-likelihood can then be used for performing likelihood ratio tests and for computing information criteria such as AIC and BIC (see the [[Model evaluation]] Section).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Monolix,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {Monolix 4.2},&lt;br /&gt;
year={2012}&lt;br /&gt;
journal = {http://www.lixoft.eu/products/monolix/product-monolix-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{comets2011package,&lt;br /&gt;
  title={saemix: Stochastic Approximation Expectation Maximization (SAEM) algorithm. R package version 0.96.1},&lt;br /&gt;
  author={Comets, E. and Lavenu, A. and Lavielle, M.},&lt;br /&gt;
journal = {http://cran.r-project.org/web/packages/saemix/index.html},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{nlmefitsa,&lt;br /&gt;
  title={nlmefitsa: fit nonlinear mixed-effects model with stochastic EM algorithm. Matlab R2013a function},&lt;br /&gt;
  author={The MathWorks},&lt;br /&gt;
journal = {http://www.mathworks.fr/fr/help/stats/nlmefitsa.html},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{beal1992nonmem,&lt;br /&gt;
  title={NONMEM users guides},&lt;br /&gt;
  author={Beal, S.L. and Sheiner, L.B. and Boeckmann, A. and Bauer, R.J.},&lt;br /&gt;
  journal={San Francisco, NONMEM Project Group, University of California},&lt;br /&gt;
  year={1992}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{pinheiro2000mixed,&lt;br /&gt;
  title={Mixed effects models in S and S-PLUS},&lt;br /&gt;
  author={Pinheiro, J.C. and Bates, D.M.},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Springer Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{pinheiro2010r,&lt;br /&gt;
  title={the R Core team (2009) nlme: Linear and Nonlinear Mixed Effects Models. R package version 3.1-96},&lt;br /&gt;
  author={Pinheiro, J. and Bates, D. and DebRoy, S. and Sarkar, D.},&lt;br /&gt;
  journal={R Foundation for Statistical Computing, Vienna},&lt;br /&gt;
  year={2010}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{spiegelhalter2003winbugs,&lt;br /&gt;
  title={WinBUGS user manual},&lt;br /&gt;
  author={Spiegelhalter, D. and Thomas, A. and Best, N. and Lunn, D.},&lt;br /&gt;
  journal={Cambridge: MRC Biostatistics Unit},&lt;br /&gt;
  year={2003}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Manual{docSPSS,&lt;br /&gt;
		title = {Linear mixed-effects modeling in SPSS. An introduction to the MIXED procedure},&lt;br /&gt;
    author = {SPSS},&lt;br /&gt;
    year = {2002},&lt;br /&gt;
    note={Technical Report}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Manual{docSAS,&lt;br /&gt;
		title = {The NLMMIXED procedure, SAS/STAT 9.2 User's Guide},&lt;br /&gt;
		chapter = {61},&lt;br /&gt;
		pages = {4337--4435},&lt;br /&gt;
		author = {SAS},&lt;br /&gt;
		year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Visualization&lt;br /&gt;
|linkNext=Model evaluation }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Estimation&amp;diff=7450</id>
		<title>Estimation</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Estimation&amp;diff=7450"/>
		<updated>2013-06-25T14:14:46Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Definitions */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;== Introduction ==&lt;br /&gt;
&lt;br /&gt;
In the modeling context, we usually assume that we have data that includes  observations $\by$,  measurement times $\bt$ and possibly  additional regression variables $\bx$.  There may also be individual covariates $\bc$, and in pharmacological applications the dose regimen $\bu$. For clarity, in the following notation we will omit  the design variables $\bt$, $\bx$ and $\bu$, and the covariates  $\bc$.&lt;br /&gt;
&lt;br /&gt;
Here, we find ourselves in the classical framework of incomplete data models. Indeed, only $\by = (y_{ij})$ is observed in the joint model $\pypsi(\by,\bpsi;\theta)$.&lt;br /&gt;
&lt;br /&gt;
Estimation tasks are common ones seen in statistics:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; Estimate the population parameter $\theta$ using the available observations and possibly  a priori information that is available.&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Evaluate the precision of the proposed estimates.&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Reconstruct missing data, here being the individual parameters $\bpsi=(\psi_i, 1\leq i \leq N)$. &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;Estimate the log-likelihood for a given model, i.e., for a given joint distribution $\qypsi$ and value of $\theta$.&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Maximum likelihood estimation of the population parameters== &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Definitions ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
''Maximum likelihood estimation''  consists of maximizing with respect to $\theta$ the ''observed likelihood''  defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\like(\theta ; \by) &amp;amp;\eqdef&amp;amp; \py(\by ; \theta) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \int \pypsi(\by,\bpsi ;\theta) \, d \bpsi .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Maximum likelihood estimation of the population parameter $\theta$ requires:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
* A model, i.e., a joint distribution $\qypsi$.  Depending on the software used, the model can be implemented using a script or a graphical user interface. $\monolix$ is extremely flexible and allows us to combine both. It is possible for instance to code the structural model using $\mlxtran$ and use the [http://en.wikipedia.org/wiki/GUI GUI] for implementing the statistical model. Whatever the options selected, the complete model can always be saved as a text file. &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* Inputs $\by$, $\bc$, $\bu$ and $\bt$.  All of these variables tend to be stored in a unique data file (see the [[Visualization#Data exploration | Data Exploration ]] Section). &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* An algorithm which allows us to maximize $\int \pypsi(\by,\bpsi ;\theta) \, d \bpsi$ with respect to $\theta$. Each software package has its own algorithms implemented. It is not our goal here to rate and compare the various algorithms and implementations. We will use exclusively the SAEM algorithm as described in [[The SAEM algorithm for estimating population parameters | The SAEM algorithm]] and implemented in $\monolix$ as we are entirely satisfied by both its theoretical and practical qualities: &amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The algorithms implemented in $\monolix$ including SAEM and its extensions ([[Mixture models|mixture models], [[Hidden Markov models|hidden Markov models]], [[Stochastic differential equations based models|SDE-based model]], [http://en.wikipedia.org/wiki/Censored_data censored data], etc.) have been published in statistical journals. Furthermore, convergence of SAEM has been rigorously proved.&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The SAEM implementation in $\monolix$ is extremely efficient for a wide variety of complex models.&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
** The SAEM implementation in $\monolix$ was done by the same group that proposed the algorithm and studied in detail its theoretical and practical properties.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= It is important to highlight the fact that for a parameter $\psi_i$ whose distribution is the transformation of a normal one (log-normal, logit-normal, etc.) the MLE $\hat{\psi}_{\rm pop}$ of the reference parameter $\psi_{\rm pop}$ is neither the mean nor the mode of the distribution. It is in fact the median.&lt;br /&gt;
&lt;br /&gt;
To show why this is the case, let $h$ be a nonlinear, twice continuously derivable and strictly increasing function such that $h(\psi_i)$ is normally distributed.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
*  First we show that it is not the mean. By definition, the MLE of $h(\psi_{\rm pop})$ is  $h(\hat{\psi}_{\rm pop})$. Thus, the estimated distribution of $h(\psi_i)$ is the normal distribution with mean $h(\hat{\psi}_{\rm pop})$, but $\esp{h(\psi_i)} = h(\hat{\psi}_{\rm pop})$ implies  that $\esp{\psi_i} \neq \hat{\psi}_{\rm pop}$ since $h$ is nonlinear. In other words, $\hat{\psi}_{\rm pop}$ is not the mean of the estimated distribution of $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Next we show that it is not the mode. Let $f$ be the pdf of $\psi_i$ and let $f_h$ be the pdf of $h(\psi_i)$. By definition, for any $h(t)\in \mathbb{R}$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
f(t) = h^\prime(t)f_h(h(t)) . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: Thus,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
f^\prime(t) = h^{\prime \prime}(t)f_h(h(t)) + h^{\prime 2}(t)f_h^\prime(h(t)) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: By definition of the mode, $f_h^\prime(h(\hat{\psi}_{\rm pop}))=0$. Since $h$ is nonlinear, $h^{\prime \prime}(\hat{\psi}_{\rm pop})\neq 0$ a.s. and  $f^\prime(\hat{\psi}_{\rm pop})\neq 0$ a.s.. In other words, $\hat{\psi}_{\rm pop}$ is not the mode of the estimated distribution of $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Now we show that it is the median. Since $h$ is a strictly increasing function,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\probs{\hat{\psi}_{\rm pop} }{\psi_i \leq \hat{\psi}_{\rm pop} } &amp;amp;=&amp;amp; \probs{\hat{\psi}_{\rm pop} }{h(\psi_i) \leq h(\hat{\psi}_{\rm pop})} \\&lt;br /&gt;
&amp;amp;=&amp;amp; 0.5 .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
: In other words, $\hat{\psi}_{\rm pop}$ is the median of the estimated distribution of $\psi_i$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Example ===&lt;br /&gt;
&lt;br /&gt;
Let us again look at the model used in the [[Visualization#Model exploration | Model Visualization]] Section. For the case of a unique dose $D$ given at time $t=0$, the structural model is written:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
ke&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; \displaystyle{\frac{D \, ka}{V(ka-ke)} }\left(e^{-ke\,t} - e^{-ka\,t} \right) \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $Cc$ is the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging). Supposing a constant error model for the concentration, the model for the observations can be easily implemented using $\mlxtran$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1est_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
parameter =  {ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
ke=Cl/V&lt;br /&gt;
Cc  = amtDose*ka/(V*(ka-ke))*(exp(-ke*t) - exp(-ka*t))&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&lt;br /&gt;
OBSERVATION:&lt;br /&gt;
Concentration = {type=continuous, prediction=Cc, errorModel=constant}&lt;br /&gt;
Hemorrhaging  = {type=event, hazard=h}&lt;br /&gt;
&lt;br /&gt;
OUTPUT:&lt;br /&gt;
output = {Concentration, Hemorrhaging}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, {{Verbatim|amtDose}} is a reserved keyword for the last  administered dose.&lt;br /&gt;
&lt;br /&gt;
The model's parameters are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$. The statistical model for the individual parameters can be defined in the $\monolix$ project file (left) and/or the $\monolix$ GUI (right):&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INDIVIDUAL:&lt;br /&gt;
     ka    = {distribution=logNormal, iiv=yes}&lt;br /&gt;
     V     = {distribution=logNormal, iiv=yes},&lt;br /&gt;
     Cl    = {distribution=normal, iiv=yes},&lt;br /&gt;
     h0    = {distribution=probitNormal, iiv=yes},&lt;br /&gt;
     gamma = {distribution=logitNormal, iiv=yes},&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=&lt;br /&gt;
[[File:Vsaem1.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Once the model is implemented, tasks such as maximum likelihood estimation can be performed using the SAEM algorithm. Certain settings in SAEM must be provided by the user. Even though SAEM is quite insensitive to the initial parameter values,&lt;br /&gt;
it is possible to perform a preliminary sensitivity analysis in order to select &amp;quot;good&amp;quot; initial values.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=Vsaem2.png|caption=Looking for good initial values for SAEM}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Then, when we run SAEM, it converges easily and quickly to the MLE:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
              parameter&lt;br /&gt;
ka          :    0.974&lt;br /&gt;
V           :     7.07&lt;br /&gt;
Cl          :     2.00&lt;br /&gt;
h0          :   0.0102&lt;br /&gt;
gamma       :    0.485&lt;br /&gt;
&lt;br /&gt;
omega_ka    :    0.668&lt;br /&gt;
omega_V     :    0.365&lt;br /&gt;
omega_Cl    :    0.588&lt;br /&gt;
omega_h0    :    0.105&lt;br /&gt;
omega_gamma :   0.0901&lt;br /&gt;
&lt;br /&gt;
a_1         :    0.345&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Parameter estimation can therefore be seen as estimating the reference values and variance of the random effects.&lt;br /&gt;
&lt;br /&gt;
In addition to these numbers, it is important to be able to graphically represent these distributions in order to see them and therefore understand them better. In effect, the interpretation of certain parameters is not always simple. Of course, we know what a normal distribution represents and in particular its mean, median and mode, which are equal (see the distribution of $Cl$ below for instance). These measures of central tendency can be different among themselves for other asymmetric distributions such as the log-normal (see the distribution of $ka$).&lt;br /&gt;
&lt;br /&gt;
Interpreting dispersion terms like  $\omega_{ka}$  and $\omega_{V}$ is not obvious either when the parameter distributions are not normal. In such cases, quartiles or quantiles of order 5% and 95% (for example) may be useful for quantitively describing the variability of these parameters.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks &lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
For a parameter $\psi$ whose distribution is log-normal, we can approximate the coefficient of variation for $\psi$ by the standard deviation $\omega_{\psi}$ of the random effect $\eta$ if this is fairly small. In effect, when $\omega_{\psi}$ is small,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\psi &amp;amp;=&amp;amp; \psi_{\rm pop} e^{\eta} \\&lt;br /&gt;
&amp;amp;\approx &amp;amp; \psi_{\rm pop}(1+ \eta) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Thus&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\esp{\psi} &amp;amp;\approx&amp;amp; \psi_{\rm pop} \\&lt;br /&gt;
\std{\psi} &amp;amp;\approx &amp;amp; \psi_{\rm pop}\omega_{\psi},&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
{\rm cv}(\psi) &amp;amp;=&amp;amp; \frac{\std{\psi} }{\esp{\psi} } \\&lt;br /&gt;
 &amp;amp;\approx &amp;amp; \omega_{\psi} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Do not forget that this approximation is only valid when $\omega$ is small and in the case of log-normal distributions. It does not carry over to any other distribution. Thus, when $\omega_{h0}=0.1$ for a probit-normal distribution or $\omega_{\gamma}=0.09$ for a logit-normal one, there is no immediate interpretation available. Only by looking at the graphical display of the pdf or by calculating some quantiles of interest can we begin to get an idea of dispersion in the parameters $h0$ and $\gamma$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=saem3b.png|caption=Estimation of the population distributions of the individual parameters of the model }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bayesian estimation of the population parameters==&lt;br /&gt;
&lt;br /&gt;
The ''Bayesian approach'' considers $\theta$ as a random vector with a ''prior distribution'' $\qth$. We can then define the posterior  distribution of $\theta$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcthy(\theta {{!}} \by ) &amp;amp;=&amp;amp; \displaystyle{ \frac{\pth( \theta )\pcyth(\by {{!}} \theta )}{\py(\by)} }\\&lt;br /&gt;
&amp;amp;=&amp;amp; \displaystyle{ \frac{\pth( \theta ) \int \pypsith(\by,\bpsi {{!}}\theta) \, d \bpsi}{\py(\by)} }.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can estimate this conditional distribution and derive any statistics (posterior mean, standard deviation, percentiles, etc.) or derive the so-called ''Maximum a Posteriori'' (MAP) estimate of $\theta$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\theta}^{\rm MAP} &amp;amp;=&amp;amp; \argmax{\theta}  \pcthy(\theta {{!}} \by ) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \argmax{\theta} \left\{  {\llike}(\theta ; \by) + \log( \pth( \theta ) ) \right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The MAP estimate  therefore maximizes a penalized version of the observed likelihood.   In other words, maximum a posteriori estimation reduces to penalized maximum likelihood estimation. Suppose for instance that $\theta$ is a scalar parameter and the prior is a normal distribution with mean $\theta_0$ and variance $\gamma^2$. Then, the MAP estimate minimizes&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hat{\theta}^{\rm MAP}  =\argmax{\theta} \left\{  {\llike} (\theta ; \by) - \displaystyle{ \frac{1}{2\gamma^2} }(\theta - \theta_0)^2 \right\} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The MAP estimate is a trade-off between the MLE which maximizes ${\llike}(\theta ; \by)$ and $\theta_0$ which minimizes $(\theta - \theta_0)^2$. The weight given to the prior directly depends on the variance of the prior distribution: the smaller $\gamma^2$ is, the closer to $\theta_0$ the MAP is. The limiting distribution considers that $\gamma^2=0$: this prior means here that $\theta$ is fixed as $\theta_0$ and no longer needs to be estimated.&lt;br /&gt;
&lt;br /&gt;
Both the Bayesian and frequentist approaches have their supporters and detractors. But rather than being dogmatic and blindly following the same rule-book every time, we need to be pragmatic and ask the right methodological questions when confronted with a new problem.&lt;br /&gt;
&lt;br /&gt;
We have to remember that Bayesian methods have been extremely successful, in particular for numerical calculations. For instance, (Bayesian) MCMC methods allow us to estimate more or less any conditional distribution coming from any hierarchical model, whereas frequentist approaches such as maximum likelihood estimation can be much more difficult to implement.&lt;br /&gt;
&lt;br /&gt;
All things said, the problem comes down to knowing whether the data contains sufficient information to answer a given question, and whether some other information may be available to help answer it. This is the essence of the art of modeling: finding the right compromise between the confidence we have in the data and prior knowledge of the problem. Each problem is different and requires a specific approach. For instance, if all the patients in a pharmacokinetic trial have essentially the same weight, it is pointless to estimate a relationship between weight and the model's PK parameters using the trial data. In this case, the modeler would be better served trying to use prior information based on physiological criteria rather than just a statistical model.&lt;br /&gt;
&lt;br /&gt;
Therefore, we can use information available to us, of course! Why not? But this information needs to be pertinent. Systematically using a prior for the parameters is not always meaningful. Can we reasonable suppose that we have access to such information? For continuous data for example, what does putting a prior on the residual error model's parameters mean in reality? A reasoned statistical approach consists of only including prior information for certain parameters (those for which we have real prior information) and having confidence in the data for the others.&lt;br /&gt;
&lt;br /&gt;
$\monolix$ allows this hybrid approach which  reconciles the Bayesian and frequentist approaches. A given parameter can be:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* a fixed constant if we have absolute confidence in its value or the data does not allow it to be estimated, essentially due to identifiability constraints.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by maximum likelihood, either because we have great confidence in the data or have no information on the parameter.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by introducing a prior and calculating the MAP estimate.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* estimated by introducing a prior and then estimating the posterior distribution.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We put aside dealing with the fixed components of $\theta$ in the following. Here are some possible situations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; ''Combined maximum likelihood and maximum a posteriori estimation'':  decompose $\theta$ into $(\theta_E,\theta_{M})$ where $\theta_E$ are the components of $\theta$ to be estimated with MLE and $\theta_{M}$ those with a prior distribution whose posterior distribution is to be maximized. Then,  $(\hat{\theta}_E , \hat{\theta}_{M} )$ below  maximizes the penalized  likelihood of $(\theta_E,\theta_{M})$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
(\hat{\theta}_E , \hat{\theta}_{M} )  &amp;amp;=&amp;amp; \argmax{\theta_E , \theta_{M} } \log(\py(\by , \theta_{M}; \theta_E)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \argmax{\theta_E , \theta_{M} }  \left\{  {\llike}(\theta_E , \theta_{M}; \by) + \log( \pth( \theta_M ) ) \right\} ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where ${\llike} (\theta_E , \theta_{M}; \by) \ \  \eqdef \ \ \log\left(\py(\by | \theta_{M}; \theta_E)\right).$&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; ''Combined maximum likelihood and posterior distribution estimation'': here, decompose  $\theta$ into $(\theta_E,\theta_{R})$ where $\theta_E$ are the components of $\theta$ to be estimated with MLE and $\theta_{R}$ those with a prior distribution whose posterior distribution is to be estimated. We  propose the following strategy for estimating $\theta_E$ and $\theta_{R}$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol style=&amp;quot;list-style-type:lower-roman&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; Compute the maximum likelihood of $\theta_E$: &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\hat{\theta}_E  &amp;amp;=&amp;amp; \argmax{\theta_E}  \log(\py(\by ; \theta_E)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \argmax{\theta_E}  \int \pmacro(\by , \theta_R ; \theta_E ) d \theta_R .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; Estimate the conditional distribution $\pmacro(\theta_{R} | \by ;\hat{\theta}_E)$. &amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is then straightforward to extend this approach to more complex situations where  some components of $\theta$ are estimated with MLE, others using MAP estimation and others still by estimating their conditional distributions.&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A PK example&lt;br /&gt;
|text=&lt;br /&gt;
In this example we use only the pharmacokinetic data and aim to estimate the population parameter distributions of the PK parameters $ka$, $V$ and $Cl$. We assume log-normal distributions for these three parameters. All of the model's population parameters are estimated by maximum likelihood estimation except $ka_{\rm pop}$ for which a log-normal distribution is used as a prior:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \log(ka_{\rm pop}) \sim {\cal N}(\log(1.5), \gamma^2) . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
$\monolix$ allows us to compute the MAP estimate and to estimate the posterior distribution of $ka_{\rm pop}$ for various values of $\gamma$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;margin-left:15%; margin-right:32%; align:center&amp;quot;&amp;gt;&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot;  align=&amp;quot;center&amp;quot; style=&amp;quot;width:100%&amp;quot;&lt;br /&gt;
{{!}} $\gamma$ {{!}}{{!}} 0 {{!}}{{!}} 0.01 {{!}}{{!}} 0.025 {{!}}{{!}} 0.05 {{!}}{{!}} 0.1 {{!}}{{!}} 0.2 {{!}}{{!}} $+ \infty$ &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}$\hat{ka}_{\rm pop}^{\rm MAP}$ {{!}}{{!}} 1.5 {{!}}{{!}} 1.49 {{!}}{{!}} 1.47 {{!}}{{!}} 1.39 {{!}}{{!}} 1.22 {{!}}{{!}} 1.11 {{!}}{{!}} 1.05 &lt;br /&gt;
{{!}}}&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=bayes1.png|caption=Prior and posterior distributions of $ka_{\rm pop}$ for different values of $\gamma$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
As expected, the posterior distribution converges to the prior distribution when the  standard deviation $\gamma$ of the prior distribution decreases. Also, the mode of the posterior distribution converges to the maximum likelihood estimate of $ka_{\rm pop}$ when $\gamma$ increases.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Estimation of the Fisher information matrix ==&lt;br /&gt;
&lt;br /&gt;
The variance of the estimator $\thmle$ and thus confidence intervals can be derived  from the [[Estimation of the observed Fisher information matrix|observed Fisher information matrix (F.I.M.)]], which itself is calculated using the observed likelihood (i.e., the pdf of the observations $\by$):&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;ofim_intro3&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\ofim(\thmle ; \by) \ \ \eqdef \ \ - \displaystyle{ \frac{\partial^2}{\partial \theta^2} }\log({\like}(\thmle ; \by)) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1)  }}&lt;br /&gt;
&lt;br /&gt;
Then, the variance-covariance matrix of the maximum likelihood estimator $\thmle$ can be estimated by the inverse of the observed F.I.M. Standard errors (s.e.) for each component of $\thmle$ are their standard deviations, i.e., the square-root of the diagonal elements of this covariance matrix. $\monolix$ also displays the (estimated) relative standard errors (r.s.e.), i.e., the (estimated) standard error divided by the value of the estimated parameter.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
              parameter     s.e. (s.a.)   r.s.e.(%)&lt;br /&gt;
ka          :    0.974         0.082           8&lt;br /&gt;
V           :     7.07          0.35           5&lt;br /&gt;
Cl          :        2          0.07           4&lt;br /&gt;
h0          :   0.0102        0.0014          14&lt;br /&gt;
gamma       :    0.485         0.015           3&lt;br /&gt;
&lt;br /&gt;
omega_ka    :    0.668         0.064          10&lt;br /&gt;
omega_V     :    0.365         0.037          10&lt;br /&gt;
omega_Cl    :    0.588         0.055           9&lt;br /&gt;
omega_h0    :    0.105         0.032          30&lt;br /&gt;
omega_gamma :   0.0901         0.044          49&lt;br /&gt;
&lt;br /&gt;
a_1         :    0.345         0.012           3&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The F.I.M. can be used for detecting overparametrization of the structural model. In effect, if the model is poorly identifiable,  certain estimators will be quite correlated and the F.I.M. will therefore be poorly conditioned and difficult to inverse. Suppose for example that we want to fit a two compartment PK model to the same data as before. The output is shown below. The large values for the relative standard errors for the inter-compartmental clearance $Q$ and the volume of the peripheral compartment $V_2$ mean that the data does not allow us to  estimate well these two parameters.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{JustCode&lt;br /&gt;
|code=&amp;lt;pre style=&amp;quot;background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;Estimation of the population parameters&lt;br /&gt;
&lt;br /&gt;
           parameter     s.e. (lin)   r.s.e.(%)&lt;br /&gt;
ka       :     0.246       0.0081             3&lt;br /&gt;
Cl       :       1.9        0.075             4&lt;br /&gt;
V1       :      1.71         0.14             8&lt;br /&gt;
Q        :  0.000171        0.024      1.43e+04&lt;br /&gt;
V2       :   0.00673          3.1      4.62e+04&lt;br /&gt;
&lt;br /&gt;
omega_ka :     0.171        0.026            15&lt;br /&gt;
omega_Cl :     0.293        0.026             9&lt;br /&gt;
omega_V1 :     0.621        0.062            10&lt;br /&gt;
omega_Q  :      5.72      1.4e+03      2.41e+04&lt;br /&gt;
omega_V2 :      4.61      1.8e+04      3.94e+05&lt;br /&gt;
&lt;br /&gt;
a        :     0.136       0.0073             5&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The Fisher information criteria is also widely used in optimal experimental design. Indeed, minimizing the variance of the estimator corresponds to maximizing the information.  Then, estimators and designs can be evaluated by looking at certain summary statistics of the covariance matrix (like the determinant or trace for instance).&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Estimation of the individual parameters ==&lt;br /&gt;
&lt;br /&gt;
Once $\theta$ has been estimated, the conditional distribution $\pmacro(\psi_i | y_i ; \hat{\theta})$ of the individual parameters $\psi_i$ can be estimated for each individual $i$ using the [[The Metropolis-Hastings algorithm for simulating the individual parameters| Metropolis-Hastings algorithm]]. For each $i$, this algorithm generates a sequence $(\psi_i^{k}, k \geq 1)$ which converges in distribution to the conditional distribution $\pmacro(\psi_i | y_i ; \hat{\theta})$  and that can be used for estimating any summary statistic of this distribution (mean, standard deviation, quantiles, etc.).&lt;br /&gt;
&lt;br /&gt;
The mode of this conditional distribution can be estimated using this sequence or by maximizing $\pmacro(\psi_i | y_i ; \hat{\theta})$ using numerical methods.&lt;br /&gt;
&lt;br /&gt;
The choice of using the conditional mean or the conditional mode is arbitrary. By default, $\monolix$ uses the conditional mode, taking the philosophy that the &amp;quot;most likely&amp;quot;  values of the individual parameters are the most suited for computing the &amp;quot;most likely&amp;quot; predictions.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=mode1.png|caption=Predicted concentrations for 6 individuals using the estimated  conditional modes of the individual PK parameters}} &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Estimation of the observed log-likelihood ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Once $\theta$ has been estimated, the observed log-likelihood of $\hat{\theta}$ is defined as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
{\llike} (\hat{\theta};\by) &amp;amp;=&amp;amp; \log({\like}(\hat{\theta};\by)) \\&lt;br /&gt;
&amp;amp;\eqdef&amp;amp; \log(\py(\by;\hat{\theta})) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The observed log-likelihood cannot be computed in closed form for nonlinear mixed effects models, but can be estimated using the methods described in the [[Estimation of the log-likelihood]] Section. The estimated  log-likelihood can then be used for performing likelihood ratio tests and for computing information criteria such as AIC and BIC (see the [[Model evaluation]] Section).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Monolix,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {Monolix 4.2},&lt;br /&gt;
year={2012}&lt;br /&gt;
journal = {http://www.lixoft.eu/products/monolix/product-monolix-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{comets2011package,&lt;br /&gt;
  title={saemix: Stochastic Approximation Expectation Maximization (SAEM) algorithm. R package version 0.96.1},&lt;br /&gt;
  author={Comets, E. and Lavenu, A. and Lavielle, M.},&lt;br /&gt;
journal = {http://cran.r-project.org/web/packages/saemix/index.html},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{nlmefitsa,&lt;br /&gt;
  title={nlmefitsa: fit nonlinear mixed-effects model with stochastic EM algorithm. Matlab R2013a function},&lt;br /&gt;
  author={The MathWorks},&lt;br /&gt;
journal = {http://www.mathworks.fr/fr/help/stats/nlmefitsa.html},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{beal1992nonmem,&lt;br /&gt;
  title={NONMEM users guides},&lt;br /&gt;
  author={Beal, S.L. and Sheiner, L.B. and Boeckmann, A. and Bauer, R.J.},&lt;br /&gt;
  journal={San Francisco, NONMEM Project Group, University of California},&lt;br /&gt;
  year={1992}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{pinheiro2000mixed,&lt;br /&gt;
  title={Mixed effects models in S and S-PLUS},&lt;br /&gt;
  author={Pinheiro, J.C. and Bates, D.M.},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Springer Verlag}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{pinheiro2010r,&lt;br /&gt;
  title={the R Core team (2009) nlme: Linear and Nonlinear Mixed Effects Models. R package version 3.1-96},&lt;br /&gt;
  author={Pinheiro, J. and Bates, D. and DebRoy, S. and Sarkar, D.},&lt;br /&gt;
  journal={R Foundation for Statistical Computing, Vienna},&lt;br /&gt;
  year={2010}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{spiegelhalter2003winbugs,&lt;br /&gt;
  title={WinBUGS user manual},&lt;br /&gt;
  author={Spiegelhalter, D. and Thomas, A. and Best, N. and Lunn, D.},&lt;br /&gt;
  journal={Cambridge: MRC Biostatistics Unit},&lt;br /&gt;
  year={2003}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Manual{docSPSS,&lt;br /&gt;
		title = {Linear mixed-effects modeling in SPSS. An introduction to the MIXED procedure},&lt;br /&gt;
    author = {SPSS},&lt;br /&gt;
    year = {2002},&lt;br /&gt;
    note={Technical Report}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Manual{docSAS,&lt;br /&gt;
		title = {The NLMMIXED procedure, SAS/STAT 9.2 User's Guide},&lt;br /&gt;
		chapter = {61},&lt;br /&gt;
		pages = {4337--4435},&lt;br /&gt;
		author = {SAS},&lt;br /&gt;
		year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Visualization&lt;br /&gt;
|linkNext=Model evaluation }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7449</id>
		<title>Visualization</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7449"/>
		<updated>2013-06-25T14:10:11Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;div style=&amp;quot;color: #2E5894; padding-left: 1.4em; padding-right:2.2em; padding-bottom:0.8em; padding:top:1&amp;quot;&amp;gt;[[Image:attention4.jpg|45px|left|link=]] &lt;br /&gt;
(If you are experiencing problems with the display of the mathematical formula, you can either try to use another browser, or use this link which should work smoothly:   http://popix.lixoft.net)&lt;br /&gt;
&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Before deciding to model data, it is very important to be able to visualize it. This is especially the case for longitudinal data when we want to see how an outcome varies with time or as a function of another outcome. We may also want to visualize how the individual covariates are distributed, visually detect if there are relationships between variables, visually compare data from different groups, etc. Development of such visual exploration tools poses no methodological problems. It is simple to write a [http://www.mathworks.fr/products/matlab/ Matlab] or [http://www.r-project.org/ R] code for one's own needs. To&lt;br /&gt;
illustrate the data visualization part of this chapter, we have created a little Matlab toolbox called [[Media:popixplore.pdf| $\popixplore$]] ({{filepath:popixplore 1.1.zip}}) which can be freely downloaded and used.&lt;br /&gt;
&lt;br /&gt;
It may also be useful to be able to visualize the model itself by undertaking a sensitivity analysis to look at how the structural model changes when we vary one or several parameters. This is important for truly understanding the structural model, i.e., what is behind the given mathematical equations. In the modeling context, we may also want to visually calibrate parameters in order to obtain predictions as close as possible to the observations. Developing such a tool is a difficult task because the tool needs to be able to easily input a model using some coding language, perform complex calculations, and provide a decent graphical interface (e.g., one that lets you easily modify the model parameters).&lt;br /&gt;
&lt;br /&gt;
Various model visualization tools exist, such as [http://www.berkeleymadonna.com/index.html Berkeley Madonna], specialized in the analysis of dynamical systems and the resolution of ordinary differential equations. Here, we use [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] for some different reasons:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
*  $\mlxplore$ uses the [http://www.lixoft.com/wp-content/resources/docs/modelMLXTRANtutorial.pdf $\mlxtran$] language which is extremely flexible and well-adapted to implementing complex mixed-effects models. Indeed, with $\mlxtran$ we can implement pharmacokinetic models with complex administration schedules, include inter-individual variability in parameters, define a statistical model for the covariates, etc. Another extremely important aspect of $\mlxtran$ is that it rigorously adopts the model representation formalisms proposed in $\wikipopix$. In other words, model implementation is completely in sync with its mathematical representation.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $\mlxplore$ provides a clear graphical interface that of course allows us to visualize the structural model, but also the statistical model, which is of fundamental importance in the population approach. We can thus visualize the impact of covariates and inter-individual variability of model parameters on predictions.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Data exploration ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The following example involves 80 individuals that receive a unique dose of an anticoagulant at time $t=0$. For each patient we then measure the plasmatic concentration of the drug at various times. This drug can cause undesirable side effects such as nose bleeds. If this happens, we also record the times at which this happens. The data is recorded in columns of a single text file {{Verbatim|pkrtte_data.csv}}. In this example, the columns are:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
'''id''' the ID number of the patient&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''time''' dose administration and observation times&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''amt''' the amount of drug administered&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''y''' the observations (concentrations and events)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''ytype''' the type of observation: 1=concentration, 2=event&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''weight''' a continuous individual covariate&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''gender''' a categorical individual covariate (F or M)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''group''' four different groups receive different doses: A=40mg, B=60mg, C=80mg, D=100mg.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata0.png|caption=The datafile {{Verbatim|pkrtte_data.csv}} }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can read this datafile with the function {{Verbatim|readdatapx}} and add  additional information about the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
datafile.name='pkrtte_data.csv';&lt;br /&gt;
datafile.format='csv';  % can be &amp;quot;csv&amp;quot;, &amp;quot;space&amp;quot;, &amp;quot;tab&amp;quot; or &amp;quot;;&amp;quot;&lt;br /&gt;
&lt;br /&gt;
info.header = {'ID','TIME','AMT','Y','YTYPE','COV','CAT','CAT'};&lt;br /&gt;
info.observation.name={'concentration','hemorrhaging'};&lt;br /&gt;
info.observation.type={'continuous','event'};&lt;br /&gt;
info.observation.unit={'mg/l',''};&lt;br /&gt;
info.covariate.unit={'kg',''};&lt;br /&gt;
info.time.unit='h';&lt;br /&gt;
&lt;br /&gt;
data=readdatapx(datafile,info);&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
How we graphically represent data depends on the type of data. Often for continuous data we use &amp;quot;spaghetti plots&amp;quot;, where all of the observations are given on the same plot, and those for each individual are joined up using line segments. Time-to-event data are usually represented using [https://en.wikipedia.org/wiki/Kaplan-Meier_survival_curve Kaplan-Meier plots], i.e., an estimate of the survival function for the first event. In the case of repeated events, we can instead represent the average cumulative number of events per individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt;&amp;gt;exploredatapx(data)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata1.png|caption=Graphical representation of the data. Left: concentrations, right: average cumulative number of events per individual}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When different groups receive different treatments, it can be useful to separately visualize  the data from each group. Here for instance we can separate the patients into groups depending on the initial dose given.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata2.png|caption=Concentration profiles per dose group}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;0&amp;quot;&lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3a.png]] &lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3b.png]]&lt;br /&gt;
|-&lt;br /&gt;
|cellspan=&amp;quot;2&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;text-align:center&amp;quot;| ''Distribution of  weight and gender per dose group'' &lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=The data file {{Verbatim|pkrtte_data.csv}} and the matlab script {{Verbatim|pkrtte_demo.m}} are available in the folder {{Verbatim|demos}} of $\popixplore$: {{filepath:popixplore 1.1.zip}}.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Model exploration==&lt;br /&gt;
&lt;br /&gt;
===Exploring the structural model===&lt;br /&gt;
&lt;br /&gt;
Suppose that we now want to visualize the following joint model which is one that can be used for simultaneously modeling  PK and time-to-event data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
k&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
\deriv{A_d} &amp;amp;=&amp;amp; - k_a \, A_d(t) \\&lt;br /&gt;
\deriv{A_c} &amp;amp;=&amp;amp;  k_a \, A_d(t) - k \, A_c(t) \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; {Ac(t)}/{V} \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $A_d$ and $A_c$ are the amounts of drug in the depot and central compartments, $Cc$ the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging for instance). The parameters of the model are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$.&lt;br /&gt;
We assume that the drug can be administered both intravenously  and orally,  meaning that the drug can be administered to both the depot and the central compartment.&lt;br /&gt;
&lt;br /&gt;
We first need to implement this model using $\mlxtran$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1_model.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
PK:&lt;br /&gt;
depot(type=1,target=Ad)&lt;br /&gt;
depot(type=2,target=Ac)&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
k = Cl/V&lt;br /&gt;
ddt_Ad = -ka*Ad&lt;br /&gt;
ddt_Ac =  ka*Ad - k*Ac&lt;br /&gt;
Cc = Ac/V&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, an administration of type 1 (resp. 2) is an oral (resp. iv) administration.&lt;br /&gt;
&lt;br /&gt;
The tasks, i.e., how the model is to be used, are then coded as an $\mlxplore$ project:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXPlore&lt;br /&gt;
|name=joint1_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint1_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
ka = 0.5&lt;br /&gt;
V = 10&lt;br /&gt;
Cl = 0.5&lt;br /&gt;
h0 = 0.01&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this example, a single dose of 50 mg is administered orally ({{Verbatim|target{{-}}Ad}} when {{Verbatim|type{{-}}1}}) at time 0. We have asked $\mlxplore$ to display the predicted concentration $Cc$ and the hazard function $h$ between $t=0$ and $t=100$ every $0.1\,h$ for a given set of parameters. We can then change the values of these parameters with the sliders to see what the impact on the two functions is.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel1.png|caption=Exploring the model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can easily modify the dose regimen without changing anything in the  model itself. Suppose for instance that we want now to compare a treatment with repeated doses of 50mg every 24 hours and a treatment with repeated doses of 25mg every 12 hours. Only the section {{Verbatim|&amp;lt;DESIGN&amp;gt;}} needs to be modified:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint2_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=0:12:144, amount=25,type=1}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=[[File:exploremodel2.png]] }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can combine different administrations (oral and intravenous for instance) into one global treatment:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint3_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=6:48:150, amount=25,type=2}&lt;br /&gt;
&lt;br /&gt;
[TREATMENT]&lt;br /&gt;
trt1={adm1, adm2}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image= [[File:exploremodel3.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
===Exploring the statistical model===&lt;br /&gt;
&lt;br /&gt;
One of the main advantages of $\mlxplore$ is its ability to  graphically display the predicted distribution of the functions of interest $Cc$ and $h$ when certain  parameters of the model are assumed to be random variables. Assume for instance that $V$, $Cl$ and $h_0$ are log-normally distributed. To take this into account, we simply need to insert a section {{Verbatim|[INDIVIDUAL]}} into the project file:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint2_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[INDIVIDUAL]&lt;br /&gt;
input={V_pop,Cl_pop,h0_pop,omega_V,omega_Cl,omega_h0}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
V =  {distribution=lognormal, reference=V_pop,  sd=omega_V}&lt;br /&gt;
Cl = {distribution=lognormal, reference=Cl_pop, sd=omega_Cl}&lt;br /&gt;
h0 = {distribution=lognormal, reference=h0_pop, sd=omega_h0}&lt;br /&gt;
&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The parameters of the model are now the population parameters $V_{\rm pop}$, $Cl_{\rm pop}$, $h0_{\rm pop}$,  $\omega_V$, $\omega_{Cl}$ and $\omega_{h_0}$ and the  parameters $k_a$ and $\gamma$ which have no inter-individual variability.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint4_project.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint2_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
V_pop = 10&lt;br /&gt;
Cl_pop = 0.5&lt;br /&gt;
h0_pop=0.01&lt;br /&gt;
omega_V = 0.2&lt;br /&gt;
omega_Cl = 0.3&lt;br /&gt;
omega_h0 = 0.2&lt;br /&gt;
ka = 0.5&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When some parameters of the model are random variables, $\mlxplore$ displays the median of the predicted distribution and several prediction intervals (the default is to use different shaded areas for the 10%, 20%, ..., 90% quantiles).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel4b.png|caption=Exploring the statistical model using $\mlxplore$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is possible to introduce covariates into the statistical model by considering for example that the volume depends on the weight, and considering that these covariates are themselves random variables. This may be important if we are for example looking to visualize the amount of variation in concentration due to variation in weight, and the variation in concentration which remains unaccounted for, caused by random effects.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel5.png|caption=Exploring the statistical model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The $\mlxtran$ model files and the $\mlxplore$ scripts can be downloaded here: {{filepath:pk mlxplore.zip}}.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{popixplore,&lt;br /&gt;
author = {POPIX Inria team},&lt;br /&gt;
title = {Popixplore 1.0},&lt;br /&gt;
url = {https://wiki.inria.fr/wikis/popix/images/7/71/Popixplore_1.1.zip},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{MLXplore,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {MLXPlore 1.0},&lt;br /&gt;
url = {http://www.lixoft.eu/products/mlxplore/mlxplore-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{macey2000berkeley,&lt;br /&gt;
  title={Berkeley Madonna user’s guide},&lt;br /&gt;
  author={Macey, R. and Oster, G. and Zahnley, T.},&lt;br /&gt;
  journal={Berkeley (CA): University of California},&lt;br /&gt;
  year={2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{chatterjee2009sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in linear regression},&lt;br /&gt;
  author={Chatterjee, S. and Hadi, A. S.},&lt;br /&gt;
  volume={327},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{sensibilité2013,&lt;br /&gt;
  title={Analyse de sensibilité et exploration de modèles},&lt;br /&gt;
  author={Faivre R. and Looss B. and  Mah&amp;amp;eacute;vas, S. and Makowski, D. and Monod, H.},&lt;br /&gt;
  year={2013},&lt;br /&gt;
  publisher={Editions Quae}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2000sensitivity,&lt;br /&gt;
  title={Sensitivity analysis},&lt;br /&gt;
  author={Saltelli, A. and Chan, K. and Scott, E. M. and others},&lt;br /&gt;
  volume={134},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Wiley New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2008global,&lt;br /&gt;
  title={Global sensitivity analysis: the primer},&lt;br /&gt;
  author={Saltelli, A. and Ratto, M. and Andres, T. and Campolongo, F. and Cariboni, J. and Gatelli, D. and Saisana, M. and Tarantola, S.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2004sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in practice: a guide to assessing scientific models},&lt;br /&gt;
  author={Saltelli, A. and Tarantola, S. and Campolongo, F. and Ratto, M.},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Next&lt;br /&gt;
|link=Modeling}}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7448</id>
		<title>Visualization</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7448"/>
		<updated>2013-06-25T14:09:04Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;div style=&amp;quot;color: #2E5894; padding-left: 1.4em; padding-right:2.2em; padding-bottom:0.8em; padding:top:1&amp;quot;&amp;gt;[[Image:attention4.jpg|45px|left|link=]] &lt;br /&gt;
(If you are experiencing problems with the display of the mathematical formula, you can either try to use another browser, or use this link which should work smoothly:   http://popix.lixoft.net)&lt;br /&gt;
&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Before deciding to model data, it is very important to be able to visualize it. This is especially the case for longitudinal data when we want to see how an outcome varies with time or as a function of another outcome. We may also want to visualize how the individual covariates are distributed, visually detect if there are relationships between variables, visually compare data from different groups, etc. Development of such visual exploration tools poses no methodological problems. It is simple to write a [http://www.mathworks.fr/products/matlab/ Matlab] or [http://www.r-project.org/ R] code for one's own needs. To&lt;br /&gt;
illustrate the data visualization part of this chapter, we have created a little Matlab toolbox called [[popixplore.pdf| $\popixplore$]] ({{filepath:popixplore 1.1.zip}}) which can be freely downloaded and used.&lt;br /&gt;
&lt;br /&gt;
It may also be useful to be able to visualize the model itself by undertaking a sensitivity analysis to look at how the structural model changes when we vary one or several parameters. This is important for truly understanding the structural model, i.e., what is behind the given mathematical equations. In the modeling context, we may also want to visually calibrate parameters in order to obtain predictions as close as possible to the observations. Developing such a tool is a difficult task because the tool needs to be able to easily input a model using some coding language, perform complex calculations, and provide a decent graphical interface (e.g., one that lets you easily modify the model parameters).&lt;br /&gt;
&lt;br /&gt;
Various model visualization tools exist, such as [http://www.berkeleymadonna.com/index.html Berkeley Madonna], specialized in the analysis of dynamical systems and the resolution of ordinary differential equations. Here, we use [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] for some different reasons:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
*  $\mlxplore$ uses the [http://www.lixoft.com/wp-content/resources/docs/modelMLXTRANtutorial.pdf $\mlxtran$] language which is extremely flexible and well-adapted to implementing complex mixed-effects models. Indeed, with $\mlxtran$ we can implement pharmacokinetic models with complex administration schedules, include inter-individual variability in parameters, define a statistical model for the covariates, etc. Another extremely important aspect of $\mlxtran$ is that it rigorously adopts the model representation formalisms proposed in $\wikipopix$. In other words, model implementation is completely in sync with its mathematical representation.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $\mlxplore$ provides a clear graphical interface that of course allows us to visualize the structural model, but also the statistical model, which is of fundamental importance in the population approach. We can thus visualize the impact of covariates and inter-individual variability of model parameters on predictions.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Data exploration ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The following example involves 80 individuals that receive a unique dose of an anticoagulant at time $t=0$. For each patient we then measure the plasmatic concentration of the drug at various times. This drug can cause undesirable side effects such as nose bleeds. If this happens, we also record the times at which this happens. The data is recorded in columns of a single text file {{Verbatim|pkrtte_data.csv}}. In this example, the columns are:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
'''id''' the ID number of the patient&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''time''' dose administration and observation times&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''amt''' the amount of drug administered&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''y''' the observations (concentrations and events)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''ytype''' the type of observation: 1=concentration, 2=event&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''weight''' a continuous individual covariate&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''gender''' a categorical individual covariate (F or M)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''group''' four different groups receive different doses: A=40mg, B=60mg, C=80mg, D=100mg.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata0.png|caption=The datafile {{Verbatim|pkrtte_data.csv}} }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can read this datafile with the function {{Verbatim|readdatapx}} and add  additional information about the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
datafile.name='pkrtte_data.csv';&lt;br /&gt;
datafile.format='csv';  % can be &amp;quot;csv&amp;quot;, &amp;quot;space&amp;quot;, &amp;quot;tab&amp;quot; or &amp;quot;;&amp;quot;&lt;br /&gt;
&lt;br /&gt;
info.header = {'ID','TIME','AMT','Y','YTYPE','COV','CAT','CAT'};&lt;br /&gt;
info.observation.name={'concentration','hemorrhaging'};&lt;br /&gt;
info.observation.type={'continuous','event'};&lt;br /&gt;
info.observation.unit={'mg/l',''};&lt;br /&gt;
info.covariate.unit={'kg',''};&lt;br /&gt;
info.time.unit='h';&lt;br /&gt;
&lt;br /&gt;
data=readdatapx(datafile,info);&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
How we graphically represent data depends on the type of data. Often for continuous data we use &amp;quot;spaghetti plots&amp;quot;, where all of the observations are given on the same plot, and those for each individual are joined up using line segments. Time-to-event data are usually represented using [https://en.wikipedia.org/wiki/Kaplan-Meier_survival_curve Kaplan-Meier plots], i.e., an estimate of the survival function for the first event. In the case of repeated events, we can instead represent the average cumulative number of events per individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt;&amp;gt;exploredatapx(data)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata1.png|caption=Graphical representation of the data. Left: concentrations, right: average cumulative number of events per individual}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When different groups receive different treatments, it can be useful to separately visualize  the data from each group. Here for instance we can separate the patients into groups depending on the initial dose given.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata2.png|caption=Concentration profiles per dose group}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;0&amp;quot;&lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3a.png]] &lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3b.png]]&lt;br /&gt;
|-&lt;br /&gt;
|cellspan=&amp;quot;2&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;text-align:center&amp;quot;| ''Distribution of  weight and gender per dose group'' &lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=The data file {{Verbatim|pkrtte_data.csv}} and the matlab script {{Verbatim|pkrtte_demo.m}} are available in the folder {{Verbatim|demos}} of $\popixplore$: {{filepath:popixplore 1.1.zip}}.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Model exploration==&lt;br /&gt;
&lt;br /&gt;
===Exploring the structural model===&lt;br /&gt;
&lt;br /&gt;
Suppose that we now want to visualize the following joint model which is one that can be used for simultaneously modeling  PK and time-to-event data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
k&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
\deriv{A_d} &amp;amp;=&amp;amp; - k_a \, A_d(t) \\&lt;br /&gt;
\deriv{A_c} &amp;amp;=&amp;amp;  k_a \, A_d(t) - k \, A_c(t) \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; {Ac(t)}/{V} \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $A_d$ and $A_c$ are the amounts of drug in the depot and central compartments, $Cc$ the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging for instance). The parameters of the model are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$.&lt;br /&gt;
We assume that the drug can be administered both intravenously  and orally,  meaning that the drug can be administered to both the depot and the central compartment.&lt;br /&gt;
&lt;br /&gt;
We first need to implement this model using $\mlxtran$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1_model.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
PK:&lt;br /&gt;
depot(type=1,target=Ad)&lt;br /&gt;
depot(type=2,target=Ac)&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
k = Cl/V&lt;br /&gt;
ddt_Ad = -ka*Ad&lt;br /&gt;
ddt_Ac =  ka*Ad - k*Ac&lt;br /&gt;
Cc = Ac/V&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, an administration of type 1 (resp. 2) is an oral (resp. iv) administration.&lt;br /&gt;
&lt;br /&gt;
The tasks, i.e., how the model is to be used, are then coded as an $\mlxplore$ project:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXPlore&lt;br /&gt;
|name=joint1_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint1_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
ka = 0.5&lt;br /&gt;
V = 10&lt;br /&gt;
Cl = 0.5&lt;br /&gt;
h0 = 0.01&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this example, a single dose of 50 mg is administered orally ({{Verbatim|target{{-}}Ad}} when {{Verbatim|type{{-}}1}}) at time 0. We have asked $\mlxplore$ to display the predicted concentration $Cc$ and the hazard function $h$ between $t=0$ and $t=100$ every $0.1\,h$ for a given set of parameters. We can then change the values of these parameters with the sliders to see what the impact on the two functions is.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel1.png|caption=Exploring the model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can easily modify the dose regimen without changing anything in the  model itself. Suppose for instance that we want now to compare a treatment with repeated doses of 50mg every 24 hours and a treatment with repeated doses of 25mg every 12 hours. Only the section {{Verbatim|&amp;lt;DESIGN&amp;gt;}} needs to be modified:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint2_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=0:12:144, amount=25,type=1}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=[[File:exploremodel2.png]] }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can combine different administrations (oral and intravenous for instance) into one global treatment:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint3_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=6:48:150, amount=25,type=2}&lt;br /&gt;
&lt;br /&gt;
[TREATMENT]&lt;br /&gt;
trt1={adm1, adm2}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image= [[File:exploremodel3.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
===Exploring the statistical model===&lt;br /&gt;
&lt;br /&gt;
One of the main advantages of $\mlxplore$ is its ability to  graphically display the predicted distribution of the functions of interest $Cc$ and $h$ when certain  parameters of the model are assumed to be random variables. Assume for instance that $V$, $Cl$ and $h_0$ are log-normally distributed. To take this into account, we simply need to insert a section {{Verbatim|[INDIVIDUAL]}} into the project file:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint2_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[INDIVIDUAL]&lt;br /&gt;
input={V_pop,Cl_pop,h0_pop,omega_V,omega_Cl,omega_h0}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
V =  {distribution=lognormal, reference=V_pop,  sd=omega_V}&lt;br /&gt;
Cl = {distribution=lognormal, reference=Cl_pop, sd=omega_Cl}&lt;br /&gt;
h0 = {distribution=lognormal, reference=h0_pop, sd=omega_h0}&lt;br /&gt;
&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The parameters of the model are now the population parameters $V_{\rm pop}$, $Cl_{\rm pop}$, $h0_{\rm pop}$,  $\omega_V$, $\omega_{Cl}$ and $\omega_{h_0}$ and the  parameters $k_a$ and $\gamma$ which have no inter-individual variability.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint4_project.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint2_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
V_pop = 10&lt;br /&gt;
Cl_pop = 0.5&lt;br /&gt;
h0_pop=0.01&lt;br /&gt;
omega_V = 0.2&lt;br /&gt;
omega_Cl = 0.3&lt;br /&gt;
omega_h0 = 0.2&lt;br /&gt;
ka = 0.5&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When some parameters of the model are random variables, $\mlxplore$ displays the median of the predicted distribution and several prediction intervals (the default is to use different shaded areas for the 10%, 20%, ..., 90% quantiles).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel4b.png|caption=Exploring the statistical model using $\mlxplore$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is possible to introduce covariates into the statistical model by considering for example that the volume depends on the weight, and considering that these covariates are themselves random variables. This may be important if we are for example looking to visualize the amount of variation in concentration due to variation in weight, and the variation in concentration which remains unaccounted for, caused by random effects.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel5.png|caption=Exploring the statistical model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The $\mlxtran$ model files and the $\mlxplore$ scripts can be downloaded here: {{filepath:pk mlxplore.zip}}.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{popixplore,&lt;br /&gt;
author = {POPIX Inria team},&lt;br /&gt;
title = {Popixplore 1.0},&lt;br /&gt;
url = {https://wiki.inria.fr/wikis/popix/images/7/71/Popixplore_1.1.zip},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{MLXplore,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {MLXPlore 1.0},&lt;br /&gt;
url = {http://www.lixoft.eu/products/mlxplore/mlxplore-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{macey2000berkeley,&lt;br /&gt;
  title={Berkeley Madonna user’s guide},&lt;br /&gt;
  author={Macey, R. and Oster, G. and Zahnley, T.},&lt;br /&gt;
  journal={Berkeley (CA): University of California},&lt;br /&gt;
  year={2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{chatterjee2009sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in linear regression},&lt;br /&gt;
  author={Chatterjee, S. and Hadi, A. S.},&lt;br /&gt;
  volume={327},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{sensibilité2013,&lt;br /&gt;
  title={Analyse de sensibilité et exploration de modèles},&lt;br /&gt;
  author={Faivre R. and Looss B. and  Mah&amp;amp;eacute;vas, S. and Makowski, D. and Monod, H.},&lt;br /&gt;
  year={2013},&lt;br /&gt;
  publisher={Editions Quae}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2000sensitivity,&lt;br /&gt;
  title={Sensitivity analysis},&lt;br /&gt;
  author={Saltelli, A. and Chan, K. and Scott, E. M. and others},&lt;br /&gt;
  volume={134},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Wiley New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2008global,&lt;br /&gt;
  title={Global sensitivity analysis: the primer},&lt;br /&gt;
  author={Saltelli, A. and Ratto, M. and Andres, T. and Campolongo, F. and Cariboni, J. and Gatelli, D. and Saisana, M. and Tarantola, S.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2004sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in practice: a guide to assessing scientific models},&lt;br /&gt;
  author={Saltelli, A. and Tarantola, S. and Campolongo, F. and Ratto, M.},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Next&lt;br /&gt;
|link=Modeling}}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7447</id>
		<title>Visualization</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7447"/>
		<updated>2013-06-25T14:08:49Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;div style=&amp;quot;color: #2E5894; padding-left: 1.4em; padding-right:2.2em; padding-bottom:0.8em; padding:top:1&amp;quot;&amp;gt;[[Image:attention4.jpg|45px|left|link=]] &lt;br /&gt;
(If you are experiencing problems with the display of the mathematical formula, you can either try to use another browser, or use this link which should work smoothly:   http://popix.lixoft.net)&lt;br /&gt;
&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Before deciding to model data, it is very important to be able to visualize it. This is especially the case for longitudinal data when we want to see how an outcome varies with time or as a function of another outcome. We may also want to visualize how the individual covariates are distributed, visually detect if there are relationships between variables, visually compare data from different groups, etc. Development of such visual exploration tools poses no methodological problems. It is simple to write a [http://www.mathworks.fr/products/matlab/ Matlab] or [http://www.r-project.org/ R] code for one's own needs. To&lt;br /&gt;
illustrate the data visualization part of this chapter, we have created a little Matlab toolbox called [[File:popixplore.pdf| $\popixplore$]] ({{filepath:popixplore 1.1.zip}}) which can be freely downloaded and used.&lt;br /&gt;
&lt;br /&gt;
It may also be useful to be able to visualize the model itself by undertaking a sensitivity analysis to look at how the structural model changes when we vary one or several parameters. This is important for truly understanding the structural model, i.e., what is behind the given mathematical equations. In the modeling context, we may also want to visually calibrate parameters in order to obtain predictions as close as possible to the observations. Developing such a tool is a difficult task because the tool needs to be able to easily input a model using some coding language, perform complex calculations, and provide a decent graphical interface (e.g., one that lets you easily modify the model parameters).&lt;br /&gt;
&lt;br /&gt;
Various model visualization tools exist, such as [http://www.berkeleymadonna.com/index.html Berkeley Madonna], specialized in the analysis of dynamical systems and the resolution of ordinary differential equations. Here, we use [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] for some different reasons:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
*  $\mlxplore$ uses the [http://www.lixoft.com/wp-content/resources/docs/modelMLXTRANtutorial.pdf $\mlxtran$] language which is extremely flexible and well-adapted to implementing complex mixed-effects models. Indeed, with $\mlxtran$ we can implement pharmacokinetic models with complex administration schedules, include inter-individual variability in parameters, define a statistical model for the covariates, etc. Another extremely important aspect of $\mlxtran$ is that it rigorously adopts the model representation formalisms proposed in $\wikipopix$. In other words, model implementation is completely in sync with its mathematical representation.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $\mlxplore$ provides a clear graphical interface that of course allows us to visualize the structural model, but also the statistical model, which is of fundamental importance in the population approach. We can thus visualize the impact of covariates and inter-individual variability of model parameters on predictions.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Data exploration ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The following example involves 80 individuals that receive a unique dose of an anticoagulant at time $t=0$. For each patient we then measure the plasmatic concentration of the drug at various times. This drug can cause undesirable side effects such as nose bleeds. If this happens, we also record the times at which this happens. The data is recorded in columns of a single text file {{Verbatim|pkrtte_data.csv}}. In this example, the columns are:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
'''id''' the ID number of the patient&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''time''' dose administration and observation times&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''amt''' the amount of drug administered&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''y''' the observations (concentrations and events)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''ytype''' the type of observation: 1=concentration, 2=event&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''weight''' a continuous individual covariate&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''gender''' a categorical individual covariate (F or M)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''group''' four different groups receive different doses: A=40mg, B=60mg, C=80mg, D=100mg.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata0.png|caption=The datafile {{Verbatim|pkrtte_data.csv}} }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can read this datafile with the function {{Verbatim|readdatapx}} and add  additional information about the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
datafile.name='pkrtte_data.csv';&lt;br /&gt;
datafile.format='csv';  % can be &amp;quot;csv&amp;quot;, &amp;quot;space&amp;quot;, &amp;quot;tab&amp;quot; or &amp;quot;;&amp;quot;&lt;br /&gt;
&lt;br /&gt;
info.header = {'ID','TIME','AMT','Y','YTYPE','COV','CAT','CAT'};&lt;br /&gt;
info.observation.name={'concentration','hemorrhaging'};&lt;br /&gt;
info.observation.type={'continuous','event'};&lt;br /&gt;
info.observation.unit={'mg/l',''};&lt;br /&gt;
info.covariate.unit={'kg',''};&lt;br /&gt;
info.time.unit='h';&lt;br /&gt;
&lt;br /&gt;
data=readdatapx(datafile,info);&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
How we graphically represent data depends on the type of data. Often for continuous data we use &amp;quot;spaghetti plots&amp;quot;, where all of the observations are given on the same plot, and those for each individual are joined up using line segments. Time-to-event data are usually represented using [https://en.wikipedia.org/wiki/Kaplan-Meier_survival_curve Kaplan-Meier plots], i.e., an estimate of the survival function for the first event. In the case of repeated events, we can instead represent the average cumulative number of events per individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt;&amp;gt;exploredatapx(data)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata1.png|caption=Graphical representation of the data. Left: concentrations, right: average cumulative number of events per individual}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When different groups receive different treatments, it can be useful to separately visualize  the data from each group. Here for instance we can separate the patients into groups depending on the initial dose given.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata2.png|caption=Concentration profiles per dose group}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;0&amp;quot;&lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3a.png]] &lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3b.png]]&lt;br /&gt;
|-&lt;br /&gt;
|cellspan=&amp;quot;2&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;text-align:center&amp;quot;| ''Distribution of  weight and gender per dose group'' &lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=The data file {{Verbatim|pkrtte_data.csv}} and the matlab script {{Verbatim|pkrtte_demo.m}} are available in the folder {{Verbatim|demos}} of $\popixplore$: {{filepath:popixplore 1.1.zip}}.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Model exploration==&lt;br /&gt;
&lt;br /&gt;
===Exploring the structural model===&lt;br /&gt;
&lt;br /&gt;
Suppose that we now want to visualize the following joint model which is one that can be used for simultaneously modeling  PK and time-to-event data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
k&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
\deriv{A_d} &amp;amp;=&amp;amp; - k_a \, A_d(t) \\&lt;br /&gt;
\deriv{A_c} &amp;amp;=&amp;amp;  k_a \, A_d(t) - k \, A_c(t) \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; {Ac(t)}/{V} \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $A_d$ and $A_c$ are the amounts of drug in the depot and central compartments, $Cc$ the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging for instance). The parameters of the model are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$.&lt;br /&gt;
We assume that the drug can be administered both intravenously  and orally,  meaning that the drug can be administered to both the depot and the central compartment.&lt;br /&gt;
&lt;br /&gt;
We first need to implement this model using $\mlxtran$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1_model.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
PK:&lt;br /&gt;
depot(type=1,target=Ad)&lt;br /&gt;
depot(type=2,target=Ac)&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
k = Cl/V&lt;br /&gt;
ddt_Ad = -ka*Ad&lt;br /&gt;
ddt_Ac =  ka*Ad - k*Ac&lt;br /&gt;
Cc = Ac/V&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, an administration of type 1 (resp. 2) is an oral (resp. iv) administration.&lt;br /&gt;
&lt;br /&gt;
The tasks, i.e., how the model is to be used, are then coded as an $\mlxplore$ project:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXPlore&lt;br /&gt;
|name=joint1_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint1_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
ka = 0.5&lt;br /&gt;
V = 10&lt;br /&gt;
Cl = 0.5&lt;br /&gt;
h0 = 0.01&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this example, a single dose of 50 mg is administered orally ({{Verbatim|target{{-}}Ad}} when {{Verbatim|type{{-}}1}}) at time 0. We have asked $\mlxplore$ to display the predicted concentration $Cc$ and the hazard function $h$ between $t=0$ and $t=100$ every $0.1\,h$ for a given set of parameters. We can then change the values of these parameters with the sliders to see what the impact on the two functions is.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel1.png|caption=Exploring the model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can easily modify the dose regimen without changing anything in the  model itself. Suppose for instance that we want now to compare a treatment with repeated doses of 50mg every 24 hours and a treatment with repeated doses of 25mg every 12 hours. Only the section {{Verbatim|&amp;lt;DESIGN&amp;gt;}} needs to be modified:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint2_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=0:12:144, amount=25,type=1}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=[[File:exploremodel2.png]] }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can combine different administrations (oral and intravenous for instance) into one global treatment:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint3_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=6:48:150, amount=25,type=2}&lt;br /&gt;
&lt;br /&gt;
[TREATMENT]&lt;br /&gt;
trt1={adm1, adm2}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image= [[File:exploremodel3.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
===Exploring the statistical model===&lt;br /&gt;
&lt;br /&gt;
One of the main advantages of $\mlxplore$ is its ability to  graphically display the predicted distribution of the functions of interest $Cc$ and $h$ when certain  parameters of the model are assumed to be random variables. Assume for instance that $V$, $Cl$ and $h_0$ are log-normally distributed. To take this into account, we simply need to insert a section {{Verbatim|[INDIVIDUAL]}} into the project file:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint2_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[INDIVIDUAL]&lt;br /&gt;
input={V_pop,Cl_pop,h0_pop,omega_V,omega_Cl,omega_h0}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
V =  {distribution=lognormal, reference=V_pop,  sd=omega_V}&lt;br /&gt;
Cl = {distribution=lognormal, reference=Cl_pop, sd=omega_Cl}&lt;br /&gt;
h0 = {distribution=lognormal, reference=h0_pop, sd=omega_h0}&lt;br /&gt;
&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The parameters of the model are now the population parameters $V_{\rm pop}$, $Cl_{\rm pop}$, $h0_{\rm pop}$,  $\omega_V$, $\omega_{Cl}$ and $\omega_{h_0}$ and the  parameters $k_a$ and $\gamma$ which have no inter-individual variability.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint4_project.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint2_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
V_pop = 10&lt;br /&gt;
Cl_pop = 0.5&lt;br /&gt;
h0_pop=0.01&lt;br /&gt;
omega_V = 0.2&lt;br /&gt;
omega_Cl = 0.3&lt;br /&gt;
omega_h0 = 0.2&lt;br /&gt;
ka = 0.5&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When some parameters of the model are random variables, $\mlxplore$ displays the median of the predicted distribution and several prediction intervals (the default is to use different shaded areas for the 10%, 20%, ..., 90% quantiles).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel4b.png|caption=Exploring the statistical model using $\mlxplore$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is possible to introduce covariates into the statistical model by considering for example that the volume depends on the weight, and considering that these covariates are themselves random variables. This may be important if we are for example looking to visualize the amount of variation in concentration due to variation in weight, and the variation in concentration which remains unaccounted for, caused by random effects.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel5.png|caption=Exploring the statistical model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The $\mlxtran$ model files and the $\mlxplore$ scripts can be downloaded here: {{filepath:pk mlxplore.zip}}.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{popixplore,&lt;br /&gt;
author = {POPIX Inria team},&lt;br /&gt;
title = {Popixplore 1.0},&lt;br /&gt;
url = {https://wiki.inria.fr/wikis/popix/images/7/71/Popixplore_1.1.zip},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{MLXplore,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {MLXPlore 1.0},&lt;br /&gt;
url = {http://www.lixoft.eu/products/mlxplore/mlxplore-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{macey2000berkeley,&lt;br /&gt;
  title={Berkeley Madonna user’s guide},&lt;br /&gt;
  author={Macey, R. and Oster, G. and Zahnley, T.},&lt;br /&gt;
  journal={Berkeley (CA): University of California},&lt;br /&gt;
  year={2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{chatterjee2009sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in linear regression},&lt;br /&gt;
  author={Chatterjee, S. and Hadi, A. S.},&lt;br /&gt;
  volume={327},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{sensibilité2013,&lt;br /&gt;
  title={Analyse de sensibilité et exploration de modèles},&lt;br /&gt;
  author={Faivre R. and Looss B. and  Mah&amp;amp;eacute;vas, S. and Makowski, D. and Monod, H.},&lt;br /&gt;
  year={2013},&lt;br /&gt;
  publisher={Editions Quae}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2000sensitivity,&lt;br /&gt;
  title={Sensitivity analysis},&lt;br /&gt;
  author={Saltelli, A. and Chan, K. and Scott, E. M. and others},&lt;br /&gt;
  volume={134},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Wiley New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2008global,&lt;br /&gt;
  title={Global sensitivity analysis: the primer},&lt;br /&gt;
  author={Saltelli, A. and Ratto, M. and Andres, T. and Campolongo, F. and Cariboni, J. and Gatelli, D. and Saisana, M. and Tarantola, S.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2004sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in practice: a guide to assessing scientific models},&lt;br /&gt;
  author={Saltelli, A. and Tarantola, S. and Campolongo, F. and Ratto, M.},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Next&lt;br /&gt;
|link=Modeling}}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=File:Popixplore.pdf&amp;diff=7446</id>
		<title>File:Popixplore.pdf</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=File:Popixplore.pdf&amp;diff=7446"/>
		<updated>2013-06-25T14:08:33Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7445</id>
		<title>Visualization</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7445"/>
		<updated>2013-06-25T14:08:13Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;div style=&amp;quot;color: #2E5894; padding-left: 1.4em; padding-right:2.2em; padding-bottom:0.8em; padding:top:1&amp;quot;&amp;gt;[[Image:attention4.jpg|45px|left|link=]] &lt;br /&gt;
(If you are experiencing problems with the display of the mathematical formula, you can either try to use another browser, or use this link which should work smoothly:   http://popix.lixoft.net)&lt;br /&gt;
&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Before deciding to model data, it is very important to be able to visualize it. This is especially the case for longitudinal data when we want to see how an outcome varies with time or as a function of another outcome. We may also want to visualize how the individual covariates are distributed, visually detect if there are relationships between variables, visually compare data from different groups, etc. Development of such visual exploration tools poses no methodological problems. It is simple to write a [http://www.mathworks.fr/products/matlab/ Matlab] or [http://www.r-project.org/ R] code for one's own needs. To&lt;br /&gt;
illustrate the data visualization part of this chapter, we have created a little Matlab toolbox called [[[[File:popixplore.pdf| $\popixplore$]] ({{filepath:popixplore 1.1.zip}}) which can be freely downloaded and used.&lt;br /&gt;
&lt;br /&gt;
It may also be useful to be able to visualize the model itself by undertaking a sensitivity analysis to look at how the structural model changes when we vary one or several parameters. This is important for truly understanding the structural model, i.e., what is behind the given mathematical equations. In the modeling context, we may also want to visually calibrate parameters in order to obtain predictions as close as possible to the observations. Developing such a tool is a difficult task because the tool needs to be able to easily input a model using some coding language, perform complex calculations, and provide a decent graphical interface (e.g., one that lets you easily modify the model parameters).&lt;br /&gt;
&lt;br /&gt;
Various model visualization tools exist, such as [http://www.berkeleymadonna.com/index.html Berkeley Madonna], specialized in the analysis of dynamical systems and the resolution of ordinary differential equations. Here, we use [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] for some different reasons:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
*  $\mlxplore$ uses the [http://www.lixoft.com/wp-content/resources/docs/modelMLXTRANtutorial.pdf $\mlxtran$] language which is extremely flexible and well-adapted to implementing complex mixed-effects models. Indeed, with $\mlxtran$ we can implement pharmacokinetic models with complex administration schedules, include inter-individual variability in parameters, define a statistical model for the covariates, etc. Another extremely important aspect of $\mlxtran$ is that it rigorously adopts the model representation formalisms proposed in $\wikipopix$. In other words, model implementation is completely in sync with its mathematical representation.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $\mlxplore$ provides a clear graphical interface that of course allows us to visualize the structural model, but also the statistical model, which is of fundamental importance in the population approach. We can thus visualize the impact of covariates and inter-individual variability of model parameters on predictions.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Data exploration ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The following example involves 80 individuals that receive a unique dose of an anticoagulant at time $t=0$. For each patient we then measure the plasmatic concentration of the drug at various times. This drug can cause undesirable side effects such as nose bleeds. If this happens, we also record the times at which this happens. The data is recorded in columns of a single text file {{Verbatim|pkrtte_data.csv}}. In this example, the columns are:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
'''id''' the ID number of the patient&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''time''' dose administration and observation times&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''amt''' the amount of drug administered&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''y''' the observations (concentrations and events)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''ytype''' the type of observation: 1=concentration, 2=event&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''weight''' a continuous individual covariate&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''gender''' a categorical individual covariate (F or M)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''group''' four different groups receive different doses: A=40mg, B=60mg, C=80mg, D=100mg.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata0.png|caption=The datafile {{Verbatim|pkrtte_data.csv}} }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can read this datafile with the function {{Verbatim|readdatapx}} and add  additional information about the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
datafile.name='pkrtte_data.csv';&lt;br /&gt;
datafile.format='csv';  % can be &amp;quot;csv&amp;quot;, &amp;quot;space&amp;quot;, &amp;quot;tab&amp;quot; or &amp;quot;;&amp;quot;&lt;br /&gt;
&lt;br /&gt;
info.header = {'ID','TIME','AMT','Y','YTYPE','COV','CAT','CAT'};&lt;br /&gt;
info.observation.name={'concentration','hemorrhaging'};&lt;br /&gt;
info.observation.type={'continuous','event'};&lt;br /&gt;
info.observation.unit={'mg/l',''};&lt;br /&gt;
info.covariate.unit={'kg',''};&lt;br /&gt;
info.time.unit='h';&lt;br /&gt;
&lt;br /&gt;
data=readdatapx(datafile,info);&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
How we graphically represent data depends on the type of data. Often for continuous data we use &amp;quot;spaghetti plots&amp;quot;, where all of the observations are given on the same plot, and those for each individual are joined up using line segments. Time-to-event data are usually represented using [https://en.wikipedia.org/wiki/Kaplan-Meier_survival_curve Kaplan-Meier plots], i.e., an estimate of the survival function for the first event. In the case of repeated events, we can instead represent the average cumulative number of events per individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt;&amp;gt;exploredatapx(data)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata1.png|caption=Graphical representation of the data. Left: concentrations, right: average cumulative number of events per individual}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When different groups receive different treatments, it can be useful to separately visualize  the data from each group. Here for instance we can separate the patients into groups depending on the initial dose given.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata2.png|caption=Concentration profiles per dose group}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;0&amp;quot;&lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3a.png]] &lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3b.png]]&lt;br /&gt;
|-&lt;br /&gt;
|cellspan=&amp;quot;2&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;text-align:center&amp;quot;| ''Distribution of  weight and gender per dose group'' &lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=The data file {{Verbatim|pkrtte_data.csv}} and the matlab script {{Verbatim|pkrtte_demo.m}} are available in the folder {{Verbatim|demos}} of $\popixplore$: {{filepath:popixplore 1.1.zip}}.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Model exploration==&lt;br /&gt;
&lt;br /&gt;
===Exploring the structural model===&lt;br /&gt;
&lt;br /&gt;
Suppose that we now want to visualize the following joint model which is one that can be used for simultaneously modeling  PK and time-to-event data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
k&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
\deriv{A_d} &amp;amp;=&amp;amp; - k_a \, A_d(t) \\&lt;br /&gt;
\deriv{A_c} &amp;amp;=&amp;amp;  k_a \, A_d(t) - k \, A_c(t) \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; {Ac(t)}/{V} \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $A_d$ and $A_c$ are the amounts of drug in the depot and central compartments, $Cc$ the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging for instance). The parameters of the model are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$.&lt;br /&gt;
We assume that the drug can be administered both intravenously  and orally,  meaning that the drug can be administered to both the depot and the central compartment.&lt;br /&gt;
&lt;br /&gt;
We first need to implement this model using $\mlxtran$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1_model.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
PK:&lt;br /&gt;
depot(type=1,target=Ad)&lt;br /&gt;
depot(type=2,target=Ac)&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
k = Cl/V&lt;br /&gt;
ddt_Ad = -ka*Ad&lt;br /&gt;
ddt_Ac =  ka*Ad - k*Ac&lt;br /&gt;
Cc = Ac/V&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, an administration of type 1 (resp. 2) is an oral (resp. iv) administration.&lt;br /&gt;
&lt;br /&gt;
The tasks, i.e., how the model is to be used, are then coded as an $\mlxplore$ project:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXPlore&lt;br /&gt;
|name=joint1_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint1_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
ka = 0.5&lt;br /&gt;
V = 10&lt;br /&gt;
Cl = 0.5&lt;br /&gt;
h0 = 0.01&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this example, a single dose of 50 mg is administered orally ({{Verbatim|target{{-}}Ad}} when {{Verbatim|type{{-}}1}}) at time 0. We have asked $\mlxplore$ to display the predicted concentration $Cc$ and the hazard function $h$ between $t=0$ and $t=100$ every $0.1\,h$ for a given set of parameters. We can then change the values of these parameters with the sliders to see what the impact on the two functions is.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel1.png|caption=Exploring the model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can easily modify the dose regimen without changing anything in the  model itself. Suppose for instance that we want now to compare a treatment with repeated doses of 50mg every 24 hours and a treatment with repeated doses of 25mg every 12 hours. Only the section {{Verbatim|&amp;lt;DESIGN&amp;gt;}} needs to be modified:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint2_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=0:12:144, amount=25,type=1}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=[[File:exploremodel2.png]] }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can combine different administrations (oral and intravenous for instance) into one global treatment:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint3_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=6:48:150, amount=25,type=2}&lt;br /&gt;
&lt;br /&gt;
[TREATMENT]&lt;br /&gt;
trt1={adm1, adm2}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image= [[File:exploremodel3.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
===Exploring the statistical model===&lt;br /&gt;
&lt;br /&gt;
One of the main advantages of $\mlxplore$ is its ability to  graphically display the predicted distribution of the functions of interest $Cc$ and $h$ when certain  parameters of the model are assumed to be random variables. Assume for instance that $V$, $Cl$ and $h_0$ are log-normally distributed. To take this into account, we simply need to insert a section {{Verbatim|[INDIVIDUAL]}} into the project file:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint2_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[INDIVIDUAL]&lt;br /&gt;
input={V_pop,Cl_pop,h0_pop,omega_V,omega_Cl,omega_h0}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
V =  {distribution=lognormal, reference=V_pop,  sd=omega_V}&lt;br /&gt;
Cl = {distribution=lognormal, reference=Cl_pop, sd=omega_Cl}&lt;br /&gt;
h0 = {distribution=lognormal, reference=h0_pop, sd=omega_h0}&lt;br /&gt;
&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The parameters of the model are now the population parameters $V_{\rm pop}$, $Cl_{\rm pop}$, $h0_{\rm pop}$,  $\omega_V$, $\omega_{Cl}$ and $\omega_{h_0}$ and the  parameters $k_a$ and $\gamma$ which have no inter-individual variability.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint4_project.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint2_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
V_pop = 10&lt;br /&gt;
Cl_pop = 0.5&lt;br /&gt;
h0_pop=0.01&lt;br /&gt;
omega_V = 0.2&lt;br /&gt;
omega_Cl = 0.3&lt;br /&gt;
omega_h0 = 0.2&lt;br /&gt;
ka = 0.5&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When some parameters of the model are random variables, $\mlxplore$ displays the median of the predicted distribution and several prediction intervals (the default is to use different shaded areas for the 10%, 20%, ..., 90% quantiles).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel4b.png|caption=Exploring the statistical model using $\mlxplore$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is possible to introduce covariates into the statistical model by considering for example that the volume depends on the weight, and considering that these covariates are themselves random variables. This may be important if we are for example looking to visualize the amount of variation in concentration due to variation in weight, and the variation in concentration which remains unaccounted for, caused by random effects.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel5.png|caption=Exploring the statistical model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The $\mlxtran$ model files and the $\mlxplore$ scripts can be downloaded here: {{filepath:pk mlxplore.zip}}.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{popixplore,&lt;br /&gt;
author = {POPIX Inria team},&lt;br /&gt;
title = {Popixplore 1.0},&lt;br /&gt;
url = {https://wiki.inria.fr/wikis/popix/images/7/71/Popixplore_1.1.zip},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{MLXplore,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {MLXPlore 1.0},&lt;br /&gt;
url = {http://www.lixoft.eu/products/mlxplore/mlxplore-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{macey2000berkeley,&lt;br /&gt;
  title={Berkeley Madonna user’s guide},&lt;br /&gt;
  author={Macey, R. and Oster, G. and Zahnley, T.},&lt;br /&gt;
  journal={Berkeley (CA): University of California},&lt;br /&gt;
  year={2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{chatterjee2009sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in linear regression},&lt;br /&gt;
  author={Chatterjee, S. and Hadi, A. S.},&lt;br /&gt;
  volume={327},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{sensibilité2013,&lt;br /&gt;
  title={Analyse de sensibilité et exploration de modèles},&lt;br /&gt;
  author={Faivre R. and Looss B. and  Mah&amp;amp;eacute;vas, S. and Makowski, D. and Monod, H.},&lt;br /&gt;
  year={2013},&lt;br /&gt;
  publisher={Editions Quae}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2000sensitivity,&lt;br /&gt;
  title={Sensitivity analysis},&lt;br /&gt;
  author={Saltelli, A. and Chan, K. and Scott, E. M. and others},&lt;br /&gt;
  volume={134},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Wiley New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2008global,&lt;br /&gt;
  title={Global sensitivity analysis: the primer},&lt;br /&gt;
  author={Saltelli, A. and Ratto, M. and Andres, T. and Campolongo, F. and Cariboni, J. and Gatelli, D. and Saisana, M. and Tarantola, S.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2004sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in practice: a guide to assessing scientific models},&lt;br /&gt;
  author={Saltelli, A. and Tarantola, S. and Campolongo, F. and Ratto, M.},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Next&lt;br /&gt;
|link=Modeling}}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7444</id>
		<title>Visualization</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7444"/>
		<updated>2013-06-25T14:02:33Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;div style=&amp;quot;color: #2E5894; padding-left: 1.4em; padding-right:2.2em; padding-bottom:0.8em; padding:top:1&amp;quot;&amp;gt;[[Image:attention4.jpg|45px|left|link=]] &lt;br /&gt;
(If you are experiencing problems with the display of the mathematical formula, you can either try to use another browser, or use this link which should work smoothly:   http://popix.lixoft.net)&lt;br /&gt;
&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Before deciding to model data, it is very important to be able to visualize it. This is especially the case for longitudinal data when we want to see how an outcome varies with time or as a function of another outcome. We may also want to visualize how the individual covariates are distributed, visually detect if there are relationships between variables, visually compare data from different groups, etc. Development of such visual exploration tools poses no methodological problems. It is simple to write a [http://www.mathworks.fr/products/matlab/ Matlab] or [http://www.r-project.org/ R] code for one's own needs. To&lt;br /&gt;
illustrate the data visualization part of this chapter, we have created a little Matlab toolbox called $\popixplore$ ({{filepath:popixplore 1.1.zip}}) which can be freely downloaded and used.&lt;br /&gt;
&lt;br /&gt;
It may also be useful to be able to visualize the model itself by undertaking a sensitivity analysis to look at how the structural model changes when we vary one or several parameters. This is important for truly understanding the structural model, i.e., what is behind the given mathematical equations. In the modeling context, we may also want to visually calibrate parameters in order to obtain predictions as close as possible to the observations. Developing such a tool is a difficult task because the tool needs to be able to easily input a model using some coding language, perform complex calculations, and provide a decent graphical interface (e.g., one that lets you easily modify the model parameters).&lt;br /&gt;
&lt;br /&gt;
Various model visualization tools exist, such as [http://www.berkeleymadonna.com/index.html Berkeley Madonna], specialized in the analysis of dynamical systems and the resolution of ordinary differential equations. Here, we use [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] for some different reasons:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
*  $\mlxplore$ uses the [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxtran$] language which is extremely flexible and well-adapted to implementing complex mixed-effects models. Indeed, with $\mlxtran$ we can implement pharmacokinetic models with complex administration schedules, include inter-individual variability in parameters, define a statistical model for the covariates, etc. Another extremely important aspect of $\mlxtran$ is that it rigorously adopts the model representation formalisms proposed in $\wikipopix$. In other words, model implementation is completely in sync with its mathematical representation.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $\mlxplore$ provides a clear graphical interface that of course allows us to visualize the structural model, but also the statistical model, which is of fundamental importance in the population approach. We can thus visualize the impact of covariates and inter-individual variability of model parameters on predictions.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Data exploration ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The following example involves 80 individuals that receive a unique dose of an anticoagulant at time $t=0$. For each patient we then measure the plasmatic concentration of the drug at various times. This drug can cause undesirable side effects such as nose bleeds. If this happens, we also record the times at which this happens. The data is recorded in columns of a single text file {{Verbatim|pkrtte_data.csv}}. In this example, the columns are:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
'''id''' the ID number of the patient&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''time''' dose administration and observation times&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''amt''' the amount of drug administered&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''y''' the observations (concentrations and events)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''ytype''' the type of observation: 1=concentration, 2=event&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''weight''' a continuous individual covariate&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''gender''' a categorical individual covariate (F or M)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''group''' four different groups receive different doses: A=40mg, B=60mg, C=80mg, D=100mg.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata0.png|caption=The datafile {{Verbatim|pkrtte_data.csv}} }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can read this datafile with the function {{Verbatim|readdatapx}} and add  additional information about the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
datafile.name='pkrtte_data.csv';&lt;br /&gt;
datafile.format='csv';  % can be &amp;quot;csv&amp;quot;, &amp;quot;space&amp;quot;, &amp;quot;tab&amp;quot; or &amp;quot;;&amp;quot;&lt;br /&gt;
&lt;br /&gt;
info.header = {'ID','TIME','AMT','Y','YTYPE','COV','CAT','CAT'};&lt;br /&gt;
info.observation.name={'concentration','hemorrhaging'};&lt;br /&gt;
info.observation.type={'continuous','event'};&lt;br /&gt;
info.observation.unit={'mg/l',''};&lt;br /&gt;
info.covariate.unit={'kg',''};&lt;br /&gt;
info.time.unit='h';&lt;br /&gt;
&lt;br /&gt;
data=readdatapx(datafile,info);&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
How we graphically represent data depends on the type of data. Often for continuous data we use &amp;quot;spaghetti plots&amp;quot;, where all of the observations are given on the same plot, and those for each individual are joined up using line segments. Time-to-event data are usually represented using [https://en.wikipedia.org/wiki/Kaplan-Meier_survival_curve Kaplan-Meier plots], i.e., an estimate of the survival function for the first event. In the case of repeated events, we can instead represent the average cumulative number of events per individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt;&amp;gt;exploredatapx(data)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata1.png|caption=Graphical representation of the data. Left: concentrations, right: average cumulative number of events per individual}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When different groups receive different treatments, it can be useful to separately visualize  the data from each group. Here for instance we can separate the patients into groups depending on the initial dose given.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata2.png|caption=Concentration profiles per dose group}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;0&amp;quot;&lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3a.png]] &lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3b.png]]&lt;br /&gt;
|-&lt;br /&gt;
|cellspan=&amp;quot;2&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;text-align:center&amp;quot;| ''Distribution of  weight and gender per dose group'' &lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=The data file {{Verbatim|pkrtte_data.csv}} and the matlab script {{Verbatim|pkrtte_demo.m}} are available in the folder {{Verbatim|demos}} of $\popixplore$: {{filepath:popixplore 1.1.zip}}.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Model exploration==&lt;br /&gt;
&lt;br /&gt;
===Exploring the structural model===&lt;br /&gt;
&lt;br /&gt;
Suppose that we now want to visualize the following joint model which is one that can be used for simultaneously modeling  PK and time-to-event data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
k&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
\deriv{A_d} &amp;amp;=&amp;amp; - k_a \, A_d(t) \\&lt;br /&gt;
\deriv{A_c} &amp;amp;=&amp;amp;  k_a \, A_d(t) - k \, A_c(t) \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; {Ac(t)}/{V} \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $A_d$ and $A_c$ are the amounts of drug in the depot and central compartments, $Cc$ the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging for instance). The parameters of the model are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$.&lt;br /&gt;
We assume that the drug can be administered both intravenously  and orally,  meaning that the drug can be administered to both the depot and the central compartment.&lt;br /&gt;
&lt;br /&gt;
We first need to implement this model using $\mlxtran$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1_model.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
PK:&lt;br /&gt;
depot(type=1,target=Ad)&lt;br /&gt;
depot(type=2,target=Ac)&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
k = Cl/V&lt;br /&gt;
ddt_Ad = -ka*Ad&lt;br /&gt;
ddt_Ac =  ka*Ad - k*Ac&lt;br /&gt;
Cc = Ac/V&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, an administration of type 1 (resp. 2) is an oral (resp. iv) administration.&lt;br /&gt;
&lt;br /&gt;
The tasks, i.e., how the model is to be used, are then coded as an [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] project:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXPlore&lt;br /&gt;
|name=joint1_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint1_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
ka = 0.5&lt;br /&gt;
V = 10&lt;br /&gt;
Cl = 0.5&lt;br /&gt;
h0 = 0.01&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this example, a single dose of 50 mg is administered orally ({{Verbatim|target{{-}}Ad}} when {{Verbatim|type{{-}}1}}) at time 0. We have asked [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] to display the predicted concentration $Cc$ and the hazard function $h$ between $t=0$ and $t=100$ every $0.1\,h$ for a given set of parameters. We can then change the values of these parameters with the sliders to see what the impact on the two functions is.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel1.png|caption=Exploring the model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can easily modify the dose regimen without changing anything in the  model itself. Suppose for instance that we want now to compare a treatment with repeated doses of 50mg every 24 hours and a treatment with repeated doses of 25mg every 12 hours. Only the section {{Verbatim|&amp;lt;DESIGN&amp;gt;}} needs to be modified:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint2_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=0:12:144, amount=25,type=1}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=[[File:exploremodel2.png]] }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can combine different administrations (oral and intravenous for instance) into one global treatment:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint3_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=6:48:150, amount=25,type=2}&lt;br /&gt;
&lt;br /&gt;
[TREATMENT]&lt;br /&gt;
trt1={adm1, adm2}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image= [[File:exploremodel3.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
===Exploring the statistical model===&lt;br /&gt;
&lt;br /&gt;
One of the main advantages of [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] is its ability to  graphically display the predicted distribution of the functions of interest $Cc$ and $h$ when certain  parameters of the model are assumed to be random variables. Assume for instance that $V$, $Cl$ and $h_0$ are log-normally distributed. To take this into account, we simply need to insert a section {{Verbatim|[INDIVIDUAL]}} into the project file:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint2_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[INDIVIDUAL]&lt;br /&gt;
input={V_pop,Cl_pop,h0_pop,omega_V,omega_Cl,omega_h0}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
V =  {distribution=lognormal, reference=V_pop,  sd=omega_V}&lt;br /&gt;
Cl = {distribution=lognormal, reference=Cl_pop, sd=omega_Cl}&lt;br /&gt;
h0 = {distribution=lognormal, reference=h0_pop, sd=omega_h0}&lt;br /&gt;
&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The parameters of the model are now the population parameters $V_{\rm pop}$, $Cl_{\rm pop}$, $h0_{\rm pop}$,  $\omega_V$, $\omega_{Cl}$ and $\omega_{h_0}$ and the  parameters $k_a$ and $\gamma$ which have no inter-individual variability.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint4_project.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint2_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
V_pop = 10&lt;br /&gt;
Cl_pop = 0.5&lt;br /&gt;
h0_pop=0.01&lt;br /&gt;
omega_V = 0.2&lt;br /&gt;
omega_Cl = 0.3&lt;br /&gt;
omega_h0 = 0.2&lt;br /&gt;
ka = 0.5&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When some parameters of the model are random variables, $\mlxplore$ displays the median of the predicted distribution and several prediction intervals (the default is to use different shaded areas for the 10%, 20%, ..., 90% quantiles).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel4b.png|caption=Exploring the statistical model using $\mlxplore$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is possible to introduce covariates into the statistical model by considering for example that the volume depends on the weight, and considering that these covariates are themselves random variables. This may be important if we are for example looking to visualize the amount of variation in concentration due to variation in weight, and the variation in concentration which remains unaccounted for, caused by random effects.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel5.png|caption=Exploring the statistical model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The $\mlxtran$ model files and the $\mlxplore$ scripts can be downloaded here: {{filepath:pk mlxplore.zip}}.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{popixplore,&lt;br /&gt;
author = {POPIX Inria team},&lt;br /&gt;
title = {Popixplore 1.0},&lt;br /&gt;
url = {https://wiki.inria.fr/wikis/popix/images/7/71/Popixplore_1.1.zip},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{MLXplore,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {MLXPlore 1.0},&lt;br /&gt;
url = {http://www.lixoft.eu/products/mlxplore/mlxplore-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{macey2000berkeley,&lt;br /&gt;
  title={Berkeley Madonna user’s guide},&lt;br /&gt;
  author={Macey, R. and Oster, G. and Zahnley, T.},&lt;br /&gt;
  journal={Berkeley (CA): University of California},&lt;br /&gt;
  year={2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{chatterjee2009sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in linear regression},&lt;br /&gt;
  author={Chatterjee, S. and Hadi, A. S.},&lt;br /&gt;
  volume={327},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{sensibilité2013,&lt;br /&gt;
  title={Analyse de sensibilité et exploration de modèles},&lt;br /&gt;
  author={Faivre R. and Looss B. and  Mah&amp;amp;eacute;vas, S. and Makowski, D. and Monod, H.},&lt;br /&gt;
  year={2013},&lt;br /&gt;
  publisher={Editions Quae}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2000sensitivity,&lt;br /&gt;
  title={Sensitivity analysis},&lt;br /&gt;
  author={Saltelli, A. and Chan, K. and Scott, E. M. and others},&lt;br /&gt;
  volume={134},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Wiley New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2008global,&lt;br /&gt;
  title={Global sensitivity analysis: the primer},&lt;br /&gt;
  author={Saltelli, A. and Ratto, M. and Andres, T. and Campolongo, F. and Cariboni, J. and Gatelli, D. and Saisana, M. and Tarantola, S.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2004sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in practice: a guide to assessing scientific models},&lt;br /&gt;
  author={Saltelli, A. and Tarantola, S. and Campolongo, F. and Ratto, M.},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Next&lt;br /&gt;
|link=Modeling}}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7443</id>
		<title>Visualization</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Visualization&amp;diff=7443"/>
		<updated>2013-06-25T14:01:51Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;div style=&amp;quot;color: #2E5894; padding-left: 1.4em; padding-right:2.2em; padding-bottom:0.8em; padding:top:1&amp;quot;&amp;gt;[[Image:attention4.jpg|45px|left|link=]] &lt;br /&gt;
(If you are experiencing problems with the display of the mathematical formula, you can either try to use another browser, or use this link which should work smoothly:   http://popix.lixoft.net)&lt;br /&gt;
&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Introduction ==&lt;br /&gt;
&lt;br /&gt;
Before deciding to model data, it is very important to be able to visualize it. This is especially the case for longitudinal data when we want to see how an outcome varies with time or as a function of another outcome. We may also want to visualize how the individual covariates are distributed, visually detect if there are relationships between variables, visually compare data from different groups, etc. Development of such visual exploration tools poses no methodological problems. It is simple to write a [http://www.mathworks.fr/products/matlab/ Matlab] or [http://www.r-project.org/ R] code for one's own needs. To&lt;br /&gt;
illustrate the data visualization part of this chapter, we have created a little Matlab toolbox called $\popixplore$ ({{filepath:popixplore 1.1.zip}}) which can be freely downloaded and used.&lt;br /&gt;
&lt;br /&gt;
It may also be useful to be able to visualize the model itself by undertaking a sensitivity analysis to look at how the structural model changes when we vary one or several parameters. This is important for truly understanding the structural model, i.e., what is behind the given mathematical equations. In the modeling context, we may also want to visually calibrate parameters in order to obtain predictions as close as possible to the observations. Developing such a tool is a difficult task because the tool needs to be able to easily input a model using some coding language, perform complex calculations, and provide a decent graphical interface (e.g., one that lets you easily modify the model parameters).&lt;br /&gt;
&lt;br /&gt;
Various model visualization tools exist, such as [http://www.berkeleymadonna.com/index.html Berkeley Madonna], specialized in the analysis of dynamical systems and the resolution of ordinary differential equations. Here, we use [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] for some different reasons:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
*  $\mlxplore$ uses the [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxtran$ language] which is extremely flexible and well-adapted to implementing complex mixed-effects models. Indeed, with $\mlxtran$ we can implement pharmacokinetic models with complex administration schedules, include inter-individual variability in parameters, define a statistical model for the covariates, etc. Another extremely important aspect of $\mlxtran$ is that it rigorously adopts the model representation formalisms proposed in $\wikipopix$. In other words, model implementation is completely in sync with its mathematical representation.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* $\mlxplore$ provides a clear graphical interface that of course allows us to visualize the structural model, but also the statistical model, which is of fundamental importance in the population approach. We can thus visualize the impact of covariates and inter-individual variability of model parameters on predictions.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Data exploration ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The following example involves 80 individuals that receive a unique dose of an anticoagulant at time $t=0$. For each patient we then measure the plasmatic concentration of the drug at various times. This drug can cause undesirable side effects such as nose bleeds. If this happens, we also record the times at which this happens. The data is recorded in columns of a single text file {{Verbatim|pkrtte_data.csv}}. In this example, the columns are:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
'''id''' the ID number of the patient&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''time''' dose administration and observation times&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''amt''' the amount of drug administered&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''y''' the observations (concentrations and events)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''ytype''' the type of observation: 1=concentration, 2=event&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''weight''' a continuous individual covariate&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''gender''' a categorical individual covariate (F or M)&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''group''' four different groups receive different doses: A=40mg, B=60mg, C=80mg, D=100mg.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata0.png|caption=The datafile {{Verbatim|pkrtte_data.csv}} }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can read this datafile with the function {{Verbatim|readdatapx}} and add  additional information about the data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
datafile.name='pkrtte_data.csv';&lt;br /&gt;
datafile.format='csv';  % can be &amp;quot;csv&amp;quot;, &amp;quot;space&amp;quot;, &amp;quot;tab&amp;quot; or &amp;quot;;&amp;quot;&lt;br /&gt;
&lt;br /&gt;
info.header = {'ID','TIME','AMT','Y','YTYPE','COV','CAT','CAT'};&lt;br /&gt;
info.observation.name={'concentration','hemorrhaging'};&lt;br /&gt;
info.observation.type={'continuous','event'};&lt;br /&gt;
info.observation.unit={'mg/l',''};&lt;br /&gt;
info.covariate.unit={'kg',''};&lt;br /&gt;
info.time.unit='h';&lt;br /&gt;
&lt;br /&gt;
data=readdatapx(datafile,info);&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
How we graphically represent data depends on the type of data. Often for continuous data we use &amp;quot;spaghetti plots&amp;quot;, where all of the observations are given on the same plot, and those for each individual are joined up using line segments. Time-to-event data are usually represented using [https://en.wikipedia.org/wiki/Kaplan-Meier_survival_curve Kaplan-Meier plots], i.e., an estimate of the survival function for the first event. In the case of repeated events, we can instead represent the average cumulative number of events per individual.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MATLABcode&lt;br /&gt;
|name=&lt;br /&gt;
|code=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;gt;&amp;gt;exploredatapx(data)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata1.png|caption=Graphical representation of the data. Left: concentrations, right: average cumulative number of events per individual}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When different groups receive different treatments, it can be useful to separately visualize  the data from each group. Here for instance we can separate the patients into groups depending on the initial dose given.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploredata2.png|caption=Concentration profiles per dose group}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;0&amp;quot;&lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3a.png]] &lt;br /&gt;
|style = &amp;quot;width:50%&amp;quot;| [[File:exploredata3b.png]]&lt;br /&gt;
|-&lt;br /&gt;
|cellspan=&amp;quot;2&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;text-align:center&amp;quot;| ''Distribution of  weight and gender per dose group'' &lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=The data file {{Verbatim|pkrtte_data.csv}} and the matlab script {{Verbatim|pkrtte_demo.m}} are available in the folder {{Verbatim|demos}} of $\popixplore$: {{filepath:popixplore 1.1.zip}}.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Model exploration==&lt;br /&gt;
&lt;br /&gt;
===Exploring the structural model===&lt;br /&gt;
&lt;br /&gt;
Suppose that we now want to visualize the following joint model which is one that can be used for simultaneously modeling  PK and time-to-event data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
k&amp;amp;=&amp;amp;Cl/V \\&lt;br /&gt;
\deriv{A_d} &amp;amp;=&amp;amp; - k_a \, A_d(t) \\&lt;br /&gt;
\deriv{A_c} &amp;amp;=&amp;amp;  k_a \, A_d(t) - k \, A_c(t) \\&lt;br /&gt;
Cc(t) &amp;amp;=&amp;amp; {Ac(t)}/{V} \\&lt;br /&gt;
h(t) &amp;amp;=&amp;amp; h_0 \, \exp(\gamma\, Cc(t)) .&lt;br /&gt;
\end{eqnarray} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, $A_d$ and $A_c$ are the amounts of drug in the depot and central compartments, $Cc$ the concentration in the central compartment and $h$ the hazard function for the event of interest (hemorrhaging for instance). The parameters of the model are the absorption rate constant $ka$, the volume of distribution $V$, the clearance $Cl$, the baseline hazard $h_0$ and the coefficient $\gamma$.&lt;br /&gt;
We assume that the drug can be administered both intravenously  and orally,  meaning that the drug can be administered to both the depot and the central compartment.&lt;br /&gt;
&lt;br /&gt;
We first need to implement this model using $\mlxtran$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint1_model.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
&lt;br /&gt;
PK:&lt;br /&gt;
depot(type=1,target=Ad)&lt;br /&gt;
depot(type=2,target=Ac)&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
k = Cl/V&lt;br /&gt;
ddt_Ad = -ka*Ad&lt;br /&gt;
ddt_Ac =  ka*Ad - k*Ac&lt;br /&gt;
Cc = Ac/V&lt;br /&gt;
h = h0*exp(gamma*Cc)&lt;br /&gt;
&amp;lt;/pre&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, an administration of type 1 (resp. 2) is an oral (resp. iv) administration.&lt;br /&gt;
&lt;br /&gt;
The tasks, i.e., how the model is to be used, are then coded as an [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] project:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXPlore&lt;br /&gt;
|name=joint1_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint1_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
ka = 0.5&lt;br /&gt;
V = 10&lt;br /&gt;
Cl = 0.5&lt;br /&gt;
h0 = 0.01&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this example, a single dose of 50 mg is administered orally ({{Verbatim|target{{-}}Ad}} when {{Verbatim|type{{-}}1}}) at time 0. We have asked [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] to display the predicted concentration $Cc$ and the hazard function $h$ between $t=0$ and $t=100$ every $0.1\,h$ for a given set of parameters. We can then change the values of these parameters with the sliders to see what the impact on the two functions is.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel1.png|caption=Exploring the model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can easily modify the dose regimen without changing anything in the  model itself. Suppose for instance that we want now to compare a treatment with repeated doses of 50mg every 24 hours and a treatment with repeated doses of 25mg every 12 hours. Only the section {{Verbatim|&amp;lt;DESIGN&amp;gt;}} needs to be modified:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint2_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=0:12:144, amount=25,type=1}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image=[[File:exploremodel2.png]] }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We can combine different administrations (oral and intravenous for instance) into one global treatment:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&amp;amp;Image&lt;br /&gt;
|title=&lt;br /&gt;
|text=&lt;br /&gt;
|code={{MLXPloreForTable&lt;br /&gt;
|name=joint3_project.txt&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0:24:144, amount=50,type=1}&lt;br /&gt;
adm2={time=6:48:150, amount=25,type=2}&lt;br /&gt;
&lt;br /&gt;
[TREATMENT]&lt;br /&gt;
trt1={adm1, adm2}&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
|image= [[File:exploremodel3.png]]&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
===Exploring the statistical model===&lt;br /&gt;
&lt;br /&gt;
One of the main advantages of [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ $\mlxplore$] is its ability to  graphically display the predicted distribution of the functions of interest $Cc$ and $h$ when certain  parameters of the model are assumed to be random variables. Assume for instance that $V$, $Cl$ and $h_0$ are log-normally distributed. To take this into account, we simply need to insert a section {{Verbatim|[INDIVIDUAL]}} into the project file:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint2_model.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
[INDIVIDUAL]&lt;br /&gt;
input={V_pop,Cl_pop,h0_pop,omega_V,omega_Cl,omega_h0}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
V =  {distribution=lognormal, reference=V_pop,  sd=omega_V}&lt;br /&gt;
Cl = {distribution=lognormal, reference=Cl_pop, sd=omega_Cl}&lt;br /&gt;
h0 = {distribution=lognormal, reference=h0_pop, sd=omega_h0}&lt;br /&gt;
&lt;br /&gt;
[PREDICTION]&lt;br /&gt;
input={ka, V, Cl, h0, gamma}&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
.&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The parameters of the model are now the population parameters $V_{\rm pop}$, $Cl_{\rm pop}$, $h0_{\rm pop}$,  $\omega_V$, $\omega_{Cl}$ and $\omega_{h_0}$ and the  parameters $k_a$ and $\gamma$ which have no inter-individual variability.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{MLXTran&lt;br /&gt;
|name=joint4_project.txt&lt;br /&gt;
|text=&amp;lt;pre style=&amp;quot;background-color: #EFEFEF; border:none&amp;quot;&amp;gt;&lt;br /&gt;
&amp;lt;MODEL&amp;gt;&lt;br /&gt;
file='joint2_model.txt'&lt;br /&gt;
&lt;br /&gt;
&amp;lt;DESIGN&amp;gt;&lt;br /&gt;
[ADMINISTRATION]&lt;br /&gt;
adm1={time=0, amount=50,type=1}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;PARAMETER&amp;gt;&lt;br /&gt;
V_pop = 10&lt;br /&gt;
Cl_pop = 0.5&lt;br /&gt;
h0_pop=0.01&lt;br /&gt;
omega_V = 0.2&lt;br /&gt;
omega_Cl = 0.3&lt;br /&gt;
omega_h0 = 0.2&lt;br /&gt;
ka = 0.5&lt;br /&gt;
gamma = 0.5&lt;br /&gt;
&lt;br /&gt;
&amp;lt;OUTPUT&amp;gt;&lt;br /&gt;
list={Cc, h}&lt;br /&gt;
grid=0:0.1:100&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
When some parameters of the model are random variables, $\mlxplore$ displays the median of the predicted distribution and several prediction intervals (the default is to use different shaded areas for the 10%, 20%, ..., 90% quantiles).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel4b.png|caption=Exploring the statistical model using $\mlxplore$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
It is possible to introduce covariates into the statistical model by considering for example that the volume depends on the weight, and considering that these covariates are themselves random variables. This may be important if we are for example looking to visualize the amount of variation in concentration due to variation in weight, and the variation in concentration which remains unaccounted for, caused by random effects.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=exploremodel5.png|caption=Exploring the statistical model using $\mlxplore$ }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The $\mlxtran$ model files and the $\mlxplore$ scripts can be downloaded here: {{filepath:pk mlxplore.zip}}.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{popixplore,&lt;br /&gt;
author = {POPIX Inria team},&lt;br /&gt;
title = {Popixplore 1.0},&lt;br /&gt;
url = {https://wiki.inria.fr/wikis/popix/images/7/71/Popixplore_1.1.zip},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@ARTICLE{MLXplore,&lt;br /&gt;
author = {Lixoft},&lt;br /&gt;
title = {MLXPlore 1.0},&lt;br /&gt;
url = {http://www.lixoft.eu/products/mlxplore/mlxplore-overview},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{macey2000berkeley,&lt;br /&gt;
  title={Berkeley Madonna user’s guide},&lt;br /&gt;
  author={Macey, R. and Oster, G. and Zahnley, T.},&lt;br /&gt;
  journal={Berkeley (CA): University of California},&lt;br /&gt;
  year={2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{chatterjee2009sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in linear regression},&lt;br /&gt;
  author={Chatterjee, S. and Hadi, A. S.},&lt;br /&gt;
  volume={327},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{sensibilité2013,&lt;br /&gt;
  title={Analyse de sensibilité et exploration de modèles},&lt;br /&gt;
  author={Faivre R. and Looss B. and  Mah&amp;amp;eacute;vas, S. and Makowski, D. and Monod, H.},&lt;br /&gt;
  year={2013},&lt;br /&gt;
  publisher={Editions Quae}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2000sensitivity,&lt;br /&gt;
  title={Sensitivity analysis},&lt;br /&gt;
  author={Saltelli, A. and Chan, K. and Scott, E. M. and others},&lt;br /&gt;
  volume={134},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Wiley New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2008global,&lt;br /&gt;
  title={Global sensitivity analysis: the primer},&lt;br /&gt;
  author={Saltelli, A. and Ratto, M. and Andres, T. and Campolongo, F. and Cariboni, J. and Gatelli, D. and Saisana, M. and Tarantola, S.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{saltelli2004sensitivity,&lt;br /&gt;
  title={Sensitivity analysis in practice: a guide to assessing scientific models},&lt;br /&gt;
  author={Saltelli, A. and Tarantola, S. and Campolongo, F. and Ratto, M.},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Next&lt;br /&gt;
|link=Modeling}}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Stochastic_differential_equations_based_models&amp;diff=7442</id>
		<title>Stochastic differential equations based models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Stochastic_differential_equations_based_models&amp;diff=7442"/>
		<updated>2013-06-25T13:57:37Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Diffusion models for dynamical systems with linear transfers */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Extensions chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Extensions]]&lt;br /&gt;
*[[Extensions| Introduction ]] | [[ Mixture models ]] | [[Hidden Markov models]]  | [[Stochastic differential equations based models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Diffusion models are known to be a relevant tool for modeling [http://en.wikipedia.org/wiki/Stochastic stochastic] dynamic phenomena, and are widely used in various fields including finance, physics, biology, physiology and control. In a population approach, a mixed-effects diffusion model  describes each individual series of observations using a system of [http://en.wikipedia.org/wiki/Stochastic_differential_equations stochastic differential equations] (SDE) while also taking into account  variability between individuals.&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity we will consider first a diffusion model for a single individual, and illustrate it with a very general [http://en.wikipedia.org/wiki/Dynamical_system dynamical system] with linear transfers and PK examples. We will then show that the extension to mixed diffusion models is fairly straightforward.&lt;br /&gt;
&lt;br /&gt;
Note that the conditional distribution $\qcypsi$  of the observations usually does not have  a closed-form expression. When the underlying system is a Gaussian linear dynamical one, the conditional pdf of the observations, $\pcypsi(y_i|\psi_i)$ can be computed using the [http://en.wikipedia.org/wiki/Kalman_filter ''Kalman filter'' (KF)]. When the system is not linear, the [http://en.wikipedia.org/wiki/Extended_Kalman_Filter ''extended Kalman filter'' (EKF)] provides an approximation of the conditional pdf.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Diffusion model==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We assume that one diffusion trajectory is  observed  with noise at discrete time points  $t_1&amp;lt;\ldots&amp;lt;t_j&amp;lt;\ldots&amp;lt;t_n$. Let us note $(X(t),t&amp;gt;0) \in \Rset^d$  the underlying dynamical process and $y_j \in \Rset$ a noisy function of $X(t_j)$, $j=1,\ldots,n$.  The general form of the diffusion model is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:SDEmodel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\left\{&lt;br /&gt;
\begin{array}{lll}&lt;br /&gt;
dX(t) &amp;amp;=&amp;amp; b(X(t),\psi)dt + \gamma(X(t),\psi)dW(t)\\[0.2cm]&lt;br /&gt;
y_{j} &amp;amp;=&amp;amp; c(X(t_{j}),\psi) + \varepsilon_{j} \\[0.2cm]&lt;br /&gt;
\varepsilon_{j} &amp;amp;\underset{i.i.d.}{\sim}&amp;amp; \mathcal{N}(0,a^2(\psi)), \quad  j=1,\ldots,n ,&lt;br /&gt;
\end{array}&lt;br /&gt;
\right. &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
with the initial condition $X(t_1) = x \in \Rset^d$. Here, $(W(t),t&amp;gt;0)$ is a standard [http://en.wikipedia.org/wiki/Wiener_process Wiener process] in $\Rset^d$ and $\varepsilon_j \in \Rset$ represents the measurement error occurring at the $j^{\mathrm{th}}$ observation, independent of $W(t)$. The measurement function $c: \ \Rset^d \times \Rset^p \rightarrow \Rset$, the drift function $b: \ \Rset^d \times \Rset^p \rightarrow \Rset^d$ and the diffusion function $\gamma: \ \Rset^d \times \Rset^p \rightarrow \mathcal{M}_d(\Rset)$, where $\mathcal{M}_d(\Rset)$ is the set of $d \times d$ matrices with real elements, are known functions that depend on an unknown parameter $\psi \in \Rset^p$.&lt;br /&gt;
&lt;br /&gt;
We can in fact consider an SDE-based model as a [http://en.wikipedia.org/wiki/Ordinary_differential_equation ODE]-based one with a stochastic component.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example:  &lt;br /&gt;
|title2= &amp;amp;#32; IV bolus with linear elimination&lt;br /&gt;
&lt;br /&gt;
|text= The ordinary differential equation &lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:ode1&amp;quot;&amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
dA_c(t) = -k A_c(t) dt&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
is usually used to describe the kinetics of a drug administered by rapid injection (IV bolus) into plasma. In bolus-specific compartmental models, plasma is treated as the single compartment  of the human body. $A_c(t)$ represents the amount of a drug ingredient in plasma at time $t$ after injection, and $k$ is the elimination  rate constant. The figure below displays the typical evolution of the amount found in the central compartment when $k=4$.&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=sde0.png|caption=Drug concentration evolution for  ODE diffusion example }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Imagine now that we aim to describe the evolution of the drug amount over time by means of stochastic differential equations rather than ordinary differential equations, in order to better describe the ''intra-individual variability'' of the observed process. We can assume for example  that the system [[#eq:ode1|(2)]] is randomly perturbed by an additive Wiener process:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:sde1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
dA_c(t) = -k A_c(t) dt + \gamma dW(t). &lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
The figure below displays four  kinetics for the amount in the central compartment, simulated from this model with $k=4$ and $\gamma=2$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=sde1.png|caption=Drug concentration evolution for SDE diffusion example }}&lt;br /&gt;
&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These kinetics are  clearly stochastic. Nevertheless, they  are not realistic because:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* they give an overly erratic description of the evolution of the drug concentration within the compartments of the human body.&lt;br /&gt;
&lt;br /&gt;
* they do not comply with certain constraints on  biological dynamics (sign, monotony).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
A more relevant model might consider that some parameters of the model randomly fluctuate over time, rather than the observed variable itself, modeling for example the elimination rate &amp;quot;constant&amp;quot; $k$ as a stochastic process $k(t)$ that randomly varies around a typical value $k^\star$.&lt;br /&gt;
&lt;br /&gt;
More generally, we can describe the fluctuations within a linear dynamical systems by considering the transfer rates, described below, as diffusion processes rather than the observed processes themselves.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Diffusion models for dynamical systems with linear transfers==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Dynamical systems have  applications in many fields. They can be used to model viral dynamics, population flows, interactions between cells, and drug pharmacokinetics. Dynamical systems involving linear transfers between different entities are usually modeled by means of a system of ODEs with the following general form:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:linearTransferODEModel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
dA(t) = K\, A(t)dt,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
where $A(t)$ is a vector whose $l^{\textrm{th}}$  component represents the condition of the $l^{\textrm{th}}$ entity at time $t$ and $K=(K_{l,l^\prime} \, 1\leq l , l^\prime \leq d)$ a deterministic matrix defined as:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:K&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
K_{l,l^\prime} = k_{l,l^\prime} &amp;amp; \textrm{if} \quad l \neq l^\prime\\&lt;br /&gt;
K_{l,l} = - k_{l,0} - \sum_{l^\prime} k_{l,l^\prime} ,&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&lt;br /&gt;
where $k_{l,l^\prime}$ represents the transfer rate from entity $l$ to entity $l^\prime$, and $k_{l,0}$ the elimination rate from entity $l$. An example of such a dynamical system with $3$ components is schematized below.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=linear.png|caption=A dynamical system with $3$ components (circles) and linear transfers between components (arrows) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this particular example, matrix $K$ would be defined as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation= &amp;lt;math&amp;gt;&lt;br /&gt;
K = \begin{pmatrix}&lt;br /&gt;
-k_{10} -k_{12} -k_{13} &amp;amp; k_{21} &amp;amp; k_{31}\\&lt;br /&gt;
k_{12} &amp;amp; -k_{20} -k_{21} -k_{23} &amp;amp; k_{32}\\&lt;br /&gt;
k_{13} &amp;amp; k_{23} &amp;amp; -k_{30} -k_{31} -k_{32}&lt;br /&gt;
\end{pmatrix}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The model defined by equations [[#eq:linearTransferODEModel|(4)]] and [[#eq:K|(5)]] is a deterministic model which assumes that transfers take place at the same rate at all times. This is  often a restrictive assumption since in reality, dynamical systems usually exhibit some random behavior. It is therefore reasonable to consider that  transfers are not constant but randomly fluctuate over  time. This new assumption leads to the following dynamical system:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:linearTransferSDEModel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
dA(t) = K(t)A(t)dt,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
where $K$ has the same structure as in [[#eq:K|(5)]] but now some components $k_{l,l^\prime}$ are stochastic processes which take non-negative values and randomly fluctuate around a typical value $k_{l,l^\prime}^\star$.&lt;br /&gt;
&lt;br /&gt;
Let us now illustrate the construction of such diffusion models using some specific examples in pharmacokinetics.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example 1: &lt;br /&gt;
|title2=  &amp;amp;#32; IV bolus administration with stochastic linear elimination&lt;br /&gt;
&lt;br /&gt;
|text= We will first extend the ODE based model defined in [[#eq:ode1|(2)]]  by assuming that $k$ is a diffusion process which takes non-negative values and  fluctuates around a typical value $k^\star$.&lt;br /&gt;
In this example, non-negativity of $k(t)$ is ensured by  defining the logarithm of the transfer rate as an [http://en.wikipedia.org/wiki/Ornstein%E2%80%93Uhlenbeck_process Ornstein-Uhlenbeck diffusion process]:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; d\log k(t)  = - \alpha \left( \log k(t) - \log k^\star  \right) dt + \gamma d W(t), &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $W$ is a standard one-dimensional Wiener process. This results in the following diffusion system:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; dX(t) = b(X(t))dt + \gamma(X(t))dW(t), &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
X(t) = \begin{pmatrix}  A_c(t) \\ \log k(t)  \end{pmatrix}, \ \ \ \&lt;br /&gt;
b(x)  = \begin{pmatrix}  -x_1 \exp(x_2) \\ -\alpha (x_2-\log k^{\star})  \end{pmatrix}, \ \ \ \&lt;br /&gt;
\gamma(x)  = \begin{pmatrix}  0 &amp;amp; 0 \\ 0 &amp;amp;  \gamma  \end{pmatrix}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Note that in this specific example, the Jacobian matrix of the drift function $b$ has a simple form: &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;  B(x)=\begin{pmatrix} - \exp(x_2) &amp;amp; -x_1 \exp(x_2)\\ 0 &amp;amp; -\alpha \end{pmatrix}. &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The two figures below display four simulated processes $k(t)$ and the associated amount processes $A_c(t)$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:sde2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
:::[[File:sde3.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We measure the concentration at times $(t_{j}, 1\leq j \leq n)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation= &amp;lt;math&amp;gt;y_j = \displaystyle{\frac{A_c(t_{j})}{V} } + a \, \teps_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The parameter vector of the model is therefore $\psi = (V, k^\star, \alpha, \gamma, a)$. We see in this example that the simulated kinetics are much more realistic than those obtained with the previous model, because:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the elimination rate process $k(t)$ is a stochastic process that takes non-negative values,&lt;br /&gt;
&lt;br /&gt;
* even though the amount process is  stochastic, it is smooth and decreases monotonically with time.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example 2: &lt;br /&gt;
|title2= &amp;amp;#32; Oral administration with first-order absorption and stochastic linear elimination&lt;br /&gt;
&lt;br /&gt;
|text=Oral PK models with first-order absorption and linear elimination are widely used to describe the time-course of a drug orally administered to a unique compartment of the human body. The drug is administrated in a depot compartment, absorbed by the central compartment with absorption rate $k_a$ and eliminated with elimination rate $k_e$. Such a model is described by the following  system of ODEs:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:oral1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\displaystyle{ \frac{d}{dt} } \begin{pmatrix} A_d(t) \\ A_c(t) \end{pmatrix} \ \ = \ \ \begin{pmatrix} -k_a &amp;amp; 0\\ k_a &amp;amp; -k_e\end{pmatrix} \begin{pmatrix} A_d(t) \\ A_c(t) \end{pmatrix},&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(7) }}&lt;br /&gt;
&lt;br /&gt;
where $A_d(t)$ and $A_c(t)$ respectively represent the amounts of drug at time $t$ in the depot  and  central compartments. Assume now that the elimination constant is driven by a stochastic process, solution to the stochastic differential equation&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; d k_e(t) = - \alpha (k_e - k_e^\star ) dt + \gamma \sqrt{k_e(t)} dW(t),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $W$ is a standard one-dimensional Wiener process. Then [[#eq:oral1|(7)]] becomes:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; dX(t) = b(X(t))dt + \gamma(X(t))dW(t). &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
X(t)= \begin{pmatrix} A_d(t) \\ A_c(t) \\ k_e(t) \end{pmatrix}, \ \ \ \&lt;br /&gt;
b(x) = \begin{pmatrix} -k_a x_1 \\ k_a x_1 -x_3 x_2 \\ -\alpha(x_3-k_e^\star ) \end{pmatrix}, \ \ \ \&lt;br /&gt;
\gamma(x) = \begin{pmatrix}  0 &amp;amp; 0 &amp;amp; 0 \\ 0 &amp;amp; 0 &amp;amp; 0\\ 0 &amp;amp; 0 &amp;amp; \gamma \sqrt{x_3}\end{pmatrix} ,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and the parameter  vector  of the model is $\psi = (V, k_a, k^\star, \alpha, \gamma, a) .$&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
In both examples, the diffusion model can be easily extended to a population approach by defining the system's parameters $\psi$ as an individual random vector.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Mixed-effects diffusion models==&lt;br /&gt;
&lt;br /&gt;
Let us now consider model [[#eq:SDEmodel|(1)]] with observations coming from several subjects. An adequate adaptation of model [[#eq:SDEmodel|(1)]] in such a context consists of considering as many dynamical systems as individuals, and defining the parameters of the individual dynamical systems as independent random variables, in such a way to correctly reflect  variability between the different trajectories. To standardize notation, we consider $N$ different subjects randomly chosen from a population and  note $n_i$  the number of observations for individual $i$, so that $t_{i1}&amp;lt;\ldots&amp;lt;t_{i,n_i}$ are subject $i$'s observation time points. $(X_i(t),t&amp;gt;0) \in \Rset^d$ and $y_{ij} \in \Rset$ will respectively denote individual $i$'s diffusion  and the observation  $X_i(t_{ij})$. The $y_{ij}$, $i=1,\ldots,N$, $j=1,\ldots,n_i$ are governed by a mixed-effects model based on a $d$-dimensional real-valued system of stochastic differential equations with the general form:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:SDEmixedModel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\left\{&lt;br /&gt;
\begin{array}{l}&lt;br /&gt;
dX_i(t) = b(X_i(t),\psi_i)dt + \gamma(X_i(t),\psi_i)dW_i(t),\\[0.2cm]&lt;br /&gt;
y_{ij} = c(X_i(t_{ij}),\psi_i) + \teps_{ij},\\[0.2cm]&lt;br /&gt;
\teps_{ij} \underset{i.i.d.}{\sim} \mathcal{N}(0,a^2(\psi_i)) \; , \; j=1,\ldots, n_i \; , \; i=1,\ldots,N,\\&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(8) }}&lt;br /&gt;
&lt;br /&gt;
with initial condition $X_i(t_1) = x_{i1} \in \Rset^d$ for $i=1,\ldots,N$. The $\psi_i$'s are unobserved independent $d$-dimensional random subject-specific parameters, drawn from a distribution $\qpsi$ which depends on a set of population parameters $\theta$, $(W_1(t),t&amp;gt;0), \ldots, (W_N(t),t&amp;gt;0)$ are standard independent Wiener processes, and the $\teps_{ij}$  are independent Gaussian random variables representing residual errors such that the $\psi_i$,  $W_i$ and  $\teps_{ij}$ are mutually independent.&lt;br /&gt;
The measurement function $c$, the drift function $b$ and the diffusion function $\gamma$ are known functions that are common to the $N$ subjects and depend on the unknown parameters $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
Assuming that the $N$ individuals are independent, the joint  pdf  is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:sdepdf&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcypsi(y_1,\ldots,y_N {{!}} \psi_1,\ldots,\psi_N) = \prod_{i=1}^{N}\pcyipsii(y_i {{!}} \psi_i).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(9) }}&lt;br /&gt;
&lt;br /&gt;
Computing the conditional distribution $\pcyipsii$ of the observations for any individual $i$ requires here to compute the conditional distribution of each observation given the past:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \pyipsiONE(y_{i1} {{!}} \psi_i)\prod_{j=2}^{n_i} p(y_{i,j} {{!}} y_{i,1},\ldots,y_{i,j-1} {{!}} \psi_i) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Except in some very specific classes of mixed-effects diffusion models, the transition density $\pmacro(y_{i,j}|y_{i,1},\ldots,y_{i,j-1} | \psi_i)$ does not have a closed-form expression since it involves the transition densities of the underlying  diffusion processes $X_i$.&lt;br /&gt;
When the underlying system is a Gaussian linear dynamical system, this density is a Gaussian density whose mean and variance can be computed using the Kalman filter. When the system is not linear, a first solution consists in approximating this density by a Gaussian density and using the extended Kalman filter for quickly computing the mean and the variance of this density. On the other hand, particle filters do not make any approximations of the transition density, but are very demanding in terms of simulation volume and computation time.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delattre2013sii,&lt;br /&gt;
  title={Coupling the SAEM algorithm and the extended Kalman filter for maximum likelihood estimation in mixed-effects diffusion models},&lt;br /&gt;
  author={Delattre, M. and Lavielle, M.},&lt;br /&gt;
  journal={Statistics and Its Interface},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Ditlevsen2005,&lt;br /&gt;
title = {Mixed Effects in Stochastic Differential Equation Models},&lt;br /&gt;
author = {Ditlevsen, S. and De Gaetano, A.},&lt;br /&gt;
journal = {REVSTAT Statistical Journal},&lt;br /&gt;
volume = {3},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {137-153}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Donnet2008,&lt;br /&gt;
title = {Parametric Inference for Mixed Models Defined by Stochastic Differential Equations},&lt;br /&gt;
author = {Donnet, S. and Samson, A.},&lt;br /&gt;
journal = {ESAIM: Probability and Statistics},&lt;br /&gt;
volume = {12},&lt;br /&gt;
year = {2008},&lt;br /&gt;
pages = {196-218}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@inproceedings{doucet2011tutorial,&lt;br /&gt;
  title={A tutorial on particle filtering and smoothing: Fifteen years later},&lt;br /&gt;
  author={Doucet, A. and Johansen, A. M.},&lt;br /&gt;
  booktitle={Oxford Handbook of Nonlinear Filtering},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  organization={Citeseer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Klim2009,&lt;br /&gt;
author = {Klim, S. and Mortensen, S. B. and Kristensen, N. R. and Overgaard, R. V. and Madsen, H.},&lt;br /&gt;
title = {Population stochastic modelling (PSM)-an R package for mixed-effects models based on stochastic differential equations},&lt;br /&gt;
journal = {Computer methods and programs in biomedicine},&lt;br /&gt;
volume = {94},&lt;br /&gt;
pages = {279-289},&lt;br /&gt;
year = {2009}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Kristensen2005,&lt;br /&gt;
title = {Using Stochastic Differential Equations for PK/PD Model Development},&lt;br /&gt;
author = {Kristensen, N. R. and Madsen, H. and Ingwersen, S. H.},&lt;br /&gt;
journal = {Journal of Pharmacokinetics and Pharmacodynamics},&lt;br /&gt;
volume = {32},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {109-141}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Mazzoni2008,&lt;br /&gt;
title = {Computational aspects of continuous-discrete extended Kalman-filtering},&lt;br /&gt;
author = {Mazzoni, T.},&lt;br /&gt;
journal = {Computational Statistics},&lt;br /&gt;
volume = {23},&lt;br /&gt;
year = {2008},&lt;br /&gt;
pages = {519-39}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{PSM,&lt;br /&gt;
title = {Population Stochastic Modelling (PSM): Model definition, description and examples},&lt;br /&gt;
author = {Mortensen, S. and Klim, S.}, &lt;br /&gt;
year = {2008},&lt;br /&gt;
url = {http://www2.imm.dtu.dk/projects/psm/},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Mortensen2007,&lt;br /&gt;
title = {A Matlab framework for estimation of NLME models using stochastic differential equations - Applications for estimation of insulin secretion rates},&lt;br /&gt;
author = {Mortensen, S. B. and Klim, S. and Dammann, B. and Kristensen, N. R.  and Madsen, H. and Overgaard, R. V.},&lt;br /&gt;
journal = {Journal of Pharmacokinetics and Pharmacodynamics},&lt;br /&gt;
volume = {34},&lt;br /&gt;
year = {2007},&lt;br /&gt;
pages = {623-642}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Overgaard2005,&lt;br /&gt;
title = {Non-Linear Mixed-Effects Models with Stochastic Differential Equations: Implementation of an Estimation Algorithm},&lt;br /&gt;
author = {Overgaard, R. V. and Jonsson, N. and Torn&amp;amp;oslash;e, C. W. and  Madsen, H.},&lt;br /&gt;
journal = {Journal of Pharmacokinetics and Pharmacodynamics},&lt;br /&gt;
volume = {32},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {85-107}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Picchini2010,&lt;br /&gt;
title = {Stochastic Differential Mixed-Effects Models},&lt;br /&gt;
author = {Picchini, U. and De Gaetano, A. and Ditlevsen, S.},&lt;br /&gt;
journal = {Scandinavian Journal of Statistics},&lt;br /&gt;
volume = {37},&lt;br /&gt;
year = {2010},&lt;br /&gt;
pages = {67-90}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Picchini2011,&lt;br /&gt;
title = {Practical Estimation of High Dimensional Stochastic Differential Mixed-Effects Models},&lt;br /&gt;
author = {Picchini, U. and Ditlevsen, S.},&lt;br /&gt;
journal = {Computational Statistics and Data Analysis},&lt;br /&gt;
volume = {55},&lt;br /&gt;
number = {3},&lt;br /&gt;
year = {2011},&lt;br /&gt;
pages = {1426-1444}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Tornoe2005,&lt;br /&gt;
title = {Stochastic Differential Equations in NONMEM: Implementation, Application, and Comparison with Ordinary Differential Equations},&lt;br /&gt;
author = {Torn&amp;amp;oslash;e, C. W. and Overgaard, R. V. and Agers&amp;amp;oslash;, H. and Nielsen, H. A. and Madsen, H. and Jonsson, E. N.},&lt;br /&gt;
journal = {Pharmaceutical Research},&lt;br /&gt;
volume = {22},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {1247-1258}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&lt;br /&gt;
|link=Hidden Markov models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Stochastic_differential_equations_based_models&amp;diff=7441</id>
		<title>Stochastic differential equations based models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Stochastic_differential_equations_based_models&amp;diff=7441"/>
		<updated>2013-06-25T13:55:18Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Extensions chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Extensions]]&lt;br /&gt;
*[[Extensions| Introduction ]] | [[ Mixture models ]] | [[Hidden Markov models]]  | [[Stochastic differential equations based models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Diffusion models are known to be a relevant tool for modeling [http://en.wikipedia.org/wiki/Stochastic stochastic] dynamic phenomena, and are widely used in various fields including finance, physics, biology, physiology and control. In a population approach, a mixed-effects diffusion model  describes each individual series of observations using a system of [http://en.wikipedia.org/wiki/Stochastic_differential_equations stochastic differential equations] (SDE) while also taking into account  variability between individuals.&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity we will consider first a diffusion model for a single individual, and illustrate it with a very general [http://en.wikipedia.org/wiki/Dynamical_system dynamical system] with linear transfers and PK examples. We will then show that the extension to mixed diffusion models is fairly straightforward.&lt;br /&gt;
&lt;br /&gt;
Note that the conditional distribution $\qcypsi$  of the observations usually does not have  a closed-form expression. When the underlying system is a Gaussian linear dynamical one, the conditional pdf of the observations, $\pcypsi(y_i|\psi_i)$ can be computed using the [http://en.wikipedia.org/wiki/Kalman_filter ''Kalman filter'' (KF)]. When the system is not linear, the [http://en.wikipedia.org/wiki/Extended_Kalman_Filter ''extended Kalman filter'' (EKF)] provides an approximation of the conditional pdf.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Diffusion model==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We assume that one diffusion trajectory is  observed  with noise at discrete time points  $t_1&amp;lt;\ldots&amp;lt;t_j&amp;lt;\ldots&amp;lt;t_n$. Let us note $(X(t),t&amp;gt;0) \in \Rset^d$  the underlying dynamical process and $y_j \in \Rset$ a noisy function of $X(t_j)$, $j=1,\ldots,n$.  The general form of the diffusion model is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:SDEmodel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\left\{&lt;br /&gt;
\begin{array}{lll}&lt;br /&gt;
dX(t) &amp;amp;=&amp;amp; b(X(t),\psi)dt + \gamma(X(t),\psi)dW(t)\\[0.2cm]&lt;br /&gt;
y_{j} &amp;amp;=&amp;amp; c(X(t_{j}),\psi) + \varepsilon_{j} \\[0.2cm]&lt;br /&gt;
\varepsilon_{j} &amp;amp;\underset{i.i.d.}{\sim}&amp;amp; \mathcal{N}(0,a^2(\psi)), \quad  j=1,\ldots,n ,&lt;br /&gt;
\end{array}&lt;br /&gt;
\right. &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
with the initial condition $X(t_1) = x \in \Rset^d$. Here, $(W(t),t&amp;gt;0)$ is a standard [http://en.wikipedia.org/wiki/Wiener_process Wiener process] in $\Rset^d$ and $\varepsilon_j \in \Rset$ represents the measurement error occurring at the $j^{\mathrm{th}}$ observation, independent of $W(t)$. The measurement function $c: \ \Rset^d \times \Rset^p \rightarrow \Rset$, the drift function $b: \ \Rset^d \times \Rset^p \rightarrow \Rset^d$ and the diffusion function $\gamma: \ \Rset^d \times \Rset^p \rightarrow \mathcal{M}_d(\Rset)$, where $\mathcal{M}_d(\Rset)$ is the set of $d \times d$ matrices with real elements, are known functions that depend on an unknown parameter $\psi \in \Rset^p$.&lt;br /&gt;
&lt;br /&gt;
We can in fact consider an SDE-based model as a [http://en.wikipedia.org/wiki/Ordinary_differential_equation ODE]-based one with a stochastic component.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example:  &lt;br /&gt;
|title2= &amp;amp;#32; IV bolus with linear elimination&lt;br /&gt;
&lt;br /&gt;
|text= The ordinary differential equation &lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:ode1&amp;quot;&amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
dA_c(t) = -k A_c(t) dt&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
is usually used to describe the kinetics of a drug administered by rapid injection (IV bolus) into plasma. In bolus-specific compartmental models, plasma is treated as the single compartment  of the human body. $A_c(t)$ represents the amount of a drug ingredient in plasma at time $t$ after injection, and $k$ is the elimination  rate constant. The figure below displays the typical evolution of the amount found in the central compartment when $k=4$.&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=sde0.png|caption=Drug concentration evolution for  ODE diffusion example }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Imagine now that we aim to describe the evolution of the drug amount over time by means of stochastic differential equations rather than ordinary differential equations, in order to better describe the ''intra-individual variability'' of the observed process. We can assume for example  that the system [[#eq:ode1|(2)]] is randomly perturbed by an additive Wiener process:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:sde1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
dA_c(t) = -k A_c(t) dt + \gamma dW(t). &lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
The figure below displays four  kinetics for the amount in the central compartment, simulated from this model with $k=4$ and $\gamma=2$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=sde1.png|caption=Drug concentration evolution for SDE diffusion example }}&lt;br /&gt;
&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These kinetics are  clearly stochastic. Nevertheless, they  are not realistic because:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* they give an overly erratic description of the evolution of the drug concentration within the compartments of the human body.&lt;br /&gt;
&lt;br /&gt;
* they do not comply with certain constraints on  biological dynamics (sign, monotony).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
A more relevant model might consider that some parameters of the model randomly fluctuate over time, rather than the observed variable itself, modeling for example the elimination rate &amp;quot;constant&amp;quot; $k$ as a stochastic process $k(t)$ that randomly varies around a typical value $k^\star$.&lt;br /&gt;
&lt;br /&gt;
More generally, we can describe the fluctuations within a linear dynamical systems by considering the transfer rates, described below, as diffusion processes rather than the observed processes themselves.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Diffusion models for dynamical systems with linear transfers==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Dynamical systems have  applications in many fields. They can be used to model viral dynamics, population flows, interactions between cells, and drug pharmacokinetics. Dynamical systems involving linear transfers between different entities are usually modeled by means of a system of ODEs with the following general form:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:linearTransferODEModel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
dA(t) = K\, A(t)dt,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
where $A(t)$ is a vector whose $l^{\textrm{th}}$  component represents the condition of the $l^{\textrm{th}}$ entity at time $t$ and $K=(K_{l,l^\prime} \, 1\leq l , l^\prime \leq d)$ a deterministic matrix defined as:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:K&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
K_{l,l^\prime} = k_{l,l^\prime} &amp;amp; \textrm{if} \quad l \neq l^\prime\\&lt;br /&gt;
K_{l,l} = - k_{l,0} - \sum_{l^\prime} k_{l,l^\prime} ,&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&lt;br /&gt;
where $k_{l,l^\prime}$ represents the transfer rate from entity $l$ to entity $l^\prime$, and $k_{l,0}$ the elimination rate from entity $l$. An example of such a dynamical system with $3$ components is schematized below.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=linear.png|caption=A dynamical system with $3$ components (circles) and linear transfers between components (arrows) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In this particular example, matrix $K$ would be defined as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation= &amp;lt;math&amp;gt;&lt;br /&gt;
K = \begin{pmatrix}&lt;br /&gt;
-k_{10} -k_{12} -k_{13} &amp;amp; k_{21} &amp;amp; k_{31}\\&lt;br /&gt;
k_{12} &amp;amp; -k_{20} -k_{21} -k_{23} &amp;amp; k_{32}\\&lt;br /&gt;
k_{13} &amp;amp; k_{23} &amp;amp; -k_{30} -k_{31} -k_{32}&lt;br /&gt;
\end{pmatrix}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The model defined by equations [[#eq:linearTransferODEModel|(4)]] and [[#eq:K|(5)]] is a deterministic model which assumes that transfers take place at the same rate at all times. This is  often a restrictive assumption since in reality, dynamical systems usually exhibit some random behavior. It is therefore reasonable to consider that  transfers are not constant but randomly fluctuate over  time. This new assumption leads to the following dynamical system:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:linearTransferSDEModel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
dA(t) = K(t)A(t)dt,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
where $K$ has the same structure as in [[#eq:K|(5)]] but now some components $k_{l,l^\prime}$ are stochastic processes which take non-negative values and randomly fluctuate around a typical value $k_{l,l^\prime}^\star$.&lt;br /&gt;
&lt;br /&gt;
Let us now illustrate the construction of such diffusion models using some specific examples in pharmacokinetics.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example 1: &lt;br /&gt;
|title2=  &amp;amp;#32; IV bolus administration with stochastic linear elimination&lt;br /&gt;
&lt;br /&gt;
|text= We will first extend the ODE based model defined in [[#eq:ode1|(2)]]  by assuming that $k$ is a diffusion process which takes non-negative values and  fluctuates around a typical value $k^\star$.&lt;br /&gt;
In this example, non-negativity of $k(t)$ is ensured by  defining the logarithm of the transfer rate as an Ornstein-Uhlenbeck diffusion process:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; d\log k(t)  = - \alpha \left( \log k(t) - \log k^\star  \right) dt + \gamma d W(t), &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $W$ is a standard one-dimensional Wiener process. This results in the following diffusion system:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; dX(t) = b(X(t))dt + \gamma(X(t))dW(t), &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
X(t) = \begin{pmatrix}  A_c(t) \\ \log k(t)  \end{pmatrix}, \ \ \ \&lt;br /&gt;
b(x)  = \begin{pmatrix}  -x_1 \exp(x_2) \\ -\alpha (x_2-\log k^{\star})  \end{pmatrix}, \ \ \ \&lt;br /&gt;
\gamma(x)  = \begin{pmatrix}  0 &amp;amp; 0 \\ 0 &amp;amp;  \gamma  \end{pmatrix}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Note that in this specific example, the Jacobian matrix of the drift function $b$ has a simple form: &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;  B(x)=\begin{pmatrix} - \exp(x_2) &amp;amp; -x_1 \exp(x_2)\\ 0 &amp;amp; -\alpha \end{pmatrix}. &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The two figures below display four simulated processes $k(t)$ and the associated amount processes $A_c(t)$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:sde2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
:::[[File:sde3.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
We measure the concentration at times $(t_{j}, 1\leq j \leq n)$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation= &amp;lt;math&amp;gt;y_j = \displaystyle{\frac{A_c(t_{j})}{V} } + a \, \teps_j . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The parameter vector of the model is therefore $\psi = (V, k^\star, \alpha, \gamma, a)$. We see in this example that the simulated kinetics are much more realistic than those obtained with the previous model, because:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the elimination rate process $k(t)$ is a stochastic process that takes non-negative values,&lt;br /&gt;
&lt;br /&gt;
* even though the amount process is  stochastic, it is smooth and decreases monotonically with time.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example 2: &lt;br /&gt;
|title2= &amp;amp;#32; Oral administration with first-order absorption and stochastic linear elimination&lt;br /&gt;
&lt;br /&gt;
|text=Oral PK models with first-order absorption and linear elimination are widely used to describe the time-course of a drug orally administered to a unique compartment of the human body. The drug is administrated in a depot compartment, absorbed by the central compartment with absorption rate $k_a$ and eliminated with elimination rate $k_e$. Such a model is described by the following  system of ODEs:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:oral1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\displaystyle{ \frac{d}{dt} } \begin{pmatrix} A_d(t) \\ A_c(t) \end{pmatrix} \ \ = \ \ \begin{pmatrix} -k_a &amp;amp; 0\\ k_a &amp;amp; -k_e\end{pmatrix} \begin{pmatrix} A_d(t) \\ A_c(t) \end{pmatrix},&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(7) }}&lt;br /&gt;
&lt;br /&gt;
where $A_d(t)$ and $A_c(t)$ respectively represent the amounts of drug at time $t$ in the depot  and  central compartments. Assume now that the elimination constant is driven by a stochastic process, solution to the stochastic differential equation&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; d k_e(t) = - \alpha (k_e - k_e^\star ) dt + \gamma \sqrt{k_e(t)} dW(t),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $W$ is a standard one-dimensional Wiener process. Then [[#eq:oral1|(7)]] becomes:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; dX(t) = b(X(t))dt + \gamma(X(t))dW(t). &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
X(t)= \begin{pmatrix} A_d(t) \\ A_c(t) \\ k_e(t) \end{pmatrix}, \ \ \ \&lt;br /&gt;
b(x) = \begin{pmatrix} -k_a x_1 \\ k_a x_1 -x_3 x_2 \\ -\alpha(x_3-k_e^\star ) \end{pmatrix}, \ \ \ \&lt;br /&gt;
\gamma(x) = \begin{pmatrix}  0 &amp;amp; 0 &amp;amp; 0 \\ 0 &amp;amp; 0 &amp;amp; 0\\ 0 &amp;amp; 0 &amp;amp; \gamma \sqrt{x_3}\end{pmatrix} ,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
and the parameter  vector  of the model is $\psi = (V, k_a, k^\star, \alpha, \gamma, a) .$&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
In both examples, the diffusion model can be easily extended to a population approach by defining the system's parameters $\psi$ as an individual random vector.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Mixed-effects diffusion models==&lt;br /&gt;
&lt;br /&gt;
Let us now consider model [[#eq:SDEmodel|(1)]] with observations coming from several subjects. An adequate adaptation of model [[#eq:SDEmodel|(1)]] in such a context consists of considering as many dynamical systems as individuals, and defining the parameters of the individual dynamical systems as independent random variables, in such a way to correctly reflect  variability between the different trajectories. To standardize notation, we consider $N$ different subjects randomly chosen from a population and  note $n_i$  the number of observations for individual $i$, so that $t_{i1}&amp;lt;\ldots&amp;lt;t_{i,n_i}$ are subject $i$'s observation time points. $(X_i(t),t&amp;gt;0) \in \Rset^d$ and $y_{ij} \in \Rset$ will respectively denote individual $i$'s diffusion  and the observation  $X_i(t_{ij})$. The $y_{ij}$, $i=1,\ldots,N$, $j=1,\ldots,n_i$ are governed by a mixed-effects model based on a $d$-dimensional real-valued system of stochastic differential equations with the general form:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:SDEmixedModel&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\left\{&lt;br /&gt;
\begin{array}{l}&lt;br /&gt;
dX_i(t) = b(X_i(t),\psi_i)dt + \gamma(X_i(t),\psi_i)dW_i(t),\\[0.2cm]&lt;br /&gt;
y_{ij} = c(X_i(t_{ij}),\psi_i) + \teps_{ij},\\[0.2cm]&lt;br /&gt;
\teps_{ij} \underset{i.i.d.}{\sim} \mathcal{N}(0,a^2(\psi_i)) \; , \; j=1,\ldots, n_i \; , \; i=1,\ldots,N,\\&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(8) }}&lt;br /&gt;
&lt;br /&gt;
with initial condition $X_i(t_1) = x_{i1} \in \Rset^d$ for $i=1,\ldots,N$. The $\psi_i$'s are unobserved independent $d$-dimensional random subject-specific parameters, drawn from a distribution $\qpsi$ which depends on a set of population parameters $\theta$, $(W_1(t),t&amp;gt;0), \ldots, (W_N(t),t&amp;gt;0)$ are standard independent Wiener processes, and the $\teps_{ij}$  are independent Gaussian random variables representing residual errors such that the $\psi_i$,  $W_i$ and  $\teps_{ij}$ are mutually independent.&lt;br /&gt;
The measurement function $c$, the drift function $b$ and the diffusion function $\gamma$ are known functions that are common to the $N$ subjects and depend on the unknown parameters $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
Assuming that the $N$ individuals are independent, the joint  pdf  is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:sdepdf&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcypsi(y_1,\ldots,y_N {{!}} \psi_1,\ldots,\psi_N) = \prod_{i=1}^{N}\pcyipsii(y_i {{!}} \psi_i).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(9) }}&lt;br /&gt;
&lt;br /&gt;
Computing the conditional distribution $\pcyipsii$ of the observations for any individual $i$ requires here to compute the conditional distribution of each observation given the past:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \pyipsiONE(y_{i1} {{!}} \psi_i)\prod_{j=2}^{n_i} p(y_{i,j} {{!}} y_{i,1},\ldots,y_{i,j-1} {{!}} \psi_i) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Except in some very specific classes of mixed-effects diffusion models, the transition density $\pmacro(y_{i,j}|y_{i,1},\ldots,y_{i,j-1} | \psi_i)$ does not have a closed-form expression since it involves the transition densities of the underlying  diffusion processes $X_i$.&lt;br /&gt;
When the underlying system is a Gaussian linear dynamical system, this density is a Gaussian density whose mean and variance can be computed using the Kalman filter. When the system is not linear, a first solution consists in approximating this density by a Gaussian density and using the extended Kalman filter for quickly computing the mean and the variance of this density. On the other hand, particle filters do not make any approximations of the transition density, but are very demanding in terms of simulation volume and computation time.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delattre2013sii,&lt;br /&gt;
  title={Coupling the SAEM algorithm and the extended Kalman filter for maximum likelihood estimation in mixed-effects diffusion models},&lt;br /&gt;
  author={Delattre, M. and Lavielle, M.},&lt;br /&gt;
  journal={Statistics and Its Interface},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Ditlevsen2005,&lt;br /&gt;
title = {Mixed Effects in Stochastic Differential Equation Models},&lt;br /&gt;
author = {Ditlevsen, S. and De Gaetano, A.},&lt;br /&gt;
journal = {REVSTAT Statistical Journal},&lt;br /&gt;
volume = {3},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {137-153}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Donnet2008,&lt;br /&gt;
title = {Parametric Inference for Mixed Models Defined by Stochastic Differential Equations},&lt;br /&gt;
author = {Donnet, S. and Samson, A.},&lt;br /&gt;
journal = {ESAIM: Probability and Statistics},&lt;br /&gt;
volume = {12},&lt;br /&gt;
year = {2008},&lt;br /&gt;
pages = {196-218}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@inproceedings{doucet2011tutorial,&lt;br /&gt;
  title={A tutorial on particle filtering and smoothing: Fifteen years later},&lt;br /&gt;
  author={Doucet, A. and Johansen, A. M.},&lt;br /&gt;
  booktitle={Oxford Handbook of Nonlinear Filtering},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  organization={Citeseer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Klim2009,&lt;br /&gt;
author = {Klim, S. and Mortensen, S. B. and Kristensen, N. R. and Overgaard, R. V. and Madsen, H.},&lt;br /&gt;
title = {Population stochastic modelling (PSM)-an R package for mixed-effects models based on stochastic differential equations},&lt;br /&gt;
journal = {Computer methods and programs in biomedicine},&lt;br /&gt;
volume = {94},&lt;br /&gt;
pages = {279-289},&lt;br /&gt;
year = {2009}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Kristensen2005,&lt;br /&gt;
title = {Using Stochastic Differential Equations for PK/PD Model Development},&lt;br /&gt;
author = {Kristensen, N. R. and Madsen, H. and Ingwersen, S. H.},&lt;br /&gt;
journal = {Journal of Pharmacokinetics and Pharmacodynamics},&lt;br /&gt;
volume = {32},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {109-141}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Mazzoni2008,&lt;br /&gt;
title = {Computational aspects of continuous-discrete extended Kalman-filtering},&lt;br /&gt;
author = {Mazzoni, T.},&lt;br /&gt;
journal = {Computational Statistics},&lt;br /&gt;
volume = {23},&lt;br /&gt;
year = {2008},&lt;br /&gt;
pages = {519-39}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{PSM,&lt;br /&gt;
title = {Population Stochastic Modelling (PSM): Model definition, description and examples},&lt;br /&gt;
author = {Mortensen, S. and Klim, S.}, &lt;br /&gt;
year = {2008},&lt;br /&gt;
url = {http://www2.imm.dtu.dk/projects/psm/},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Mortensen2007,&lt;br /&gt;
title = {A Matlab framework for estimation of NLME models using stochastic differential equations - Applications for estimation of insulin secretion rates},&lt;br /&gt;
author = {Mortensen, S. B. and Klim, S. and Dammann, B. and Kristensen, N. R.  and Madsen, H. and Overgaard, R. V.},&lt;br /&gt;
journal = {Journal of Pharmacokinetics and Pharmacodynamics},&lt;br /&gt;
volume = {34},&lt;br /&gt;
year = {2007},&lt;br /&gt;
pages = {623-642}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Overgaard2005,&lt;br /&gt;
title = {Non-Linear Mixed-Effects Models with Stochastic Differential Equations: Implementation of an Estimation Algorithm},&lt;br /&gt;
author = {Overgaard, R. V. and Jonsson, N. and Torn&amp;amp;oslash;e, C. W. and  Madsen, H.},&lt;br /&gt;
journal = {Journal of Pharmacokinetics and Pharmacodynamics},&lt;br /&gt;
volume = {32},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {85-107}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Picchini2010,&lt;br /&gt;
title = {Stochastic Differential Mixed-Effects Models},&lt;br /&gt;
author = {Picchini, U. and De Gaetano, A. and Ditlevsen, S.},&lt;br /&gt;
journal = {Scandinavian Journal of Statistics},&lt;br /&gt;
volume = {37},&lt;br /&gt;
year = {2010},&lt;br /&gt;
pages = {67-90}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Picchini2011,&lt;br /&gt;
title = {Practical Estimation of High Dimensional Stochastic Differential Mixed-Effects Models},&lt;br /&gt;
author = {Picchini, U. and Ditlevsen, S.},&lt;br /&gt;
journal = {Computational Statistics and Data Analysis},&lt;br /&gt;
volume = {55},&lt;br /&gt;
number = {3},&lt;br /&gt;
year = {2011},&lt;br /&gt;
pages = {1426-1444}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@Article{Tornoe2005,&lt;br /&gt;
title = {Stochastic Differential Equations in NONMEM: Implementation, Application, and Comparison with Ordinary Differential Equations},&lt;br /&gt;
author = {Torn&amp;amp;oslash;e, C. W. and Overgaard, R. V. and Agers&amp;amp;oslash;, H. and Nielsen, H. A. and Madsen, H. and Jonsson, E. N.},&lt;br /&gt;
journal = {Pharmaceutical Research},&lt;br /&gt;
volume = {22},&lt;br /&gt;
year = {2005},&lt;br /&gt;
pages = {1247-1258}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&lt;br /&gt;
|link=Hidden Markov models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Hidden_Markov_models&amp;diff=7440</id>
		<title>Hidden Markov models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Hidden_Markov_models&amp;diff=7440"/>
		<updated>2013-06-25T13:44:54Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Mixed hidden Markov models */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Extensions chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Extensions]]&lt;br /&gt;
*[[Extensions| Introduction ]] | [[ Mixture models ]] | [[Hidden Markov models]]  | [[Stochastic differential equations based models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
[http://en.wikipedia.org/wiki/Markov_chain Markov chains] are a useful tool for analyzing categorical longitudinal data. However, sometimes the [https://en.wikipedia.org/wiki/Markov_process Markov process]  cannot be directly observed, though some output, dependent on the&lt;br /&gt;
(hidden) state, is visible. More precisely, we assume that the distribution of this  observable output depends on the underlying hidden state. Such models are called hidden Markov models (HMMs).&lt;br /&gt;
HMMs can be applied in many contexts and have turned out to be particularly pertinent in several biological contexts. For example, they are useful when characterizing diseases for which the existence of several discrete stages of illness is a realistic assumption, e.g., epilepsy and migraines.&lt;br /&gt;
&lt;br /&gt;
Here, we will consider a parametric framework with Markov chains in a discrete and finite state space $\mathbf{K} = \{1,\ldots,K\}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Mixed hidden Markov models==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
HMMs have been developed to describe how a given system moves from one state to another over time, in situations where the successive visited states are unknown and a set of observations is the only available information to describe the dynamics of the system. HMMs can be seen as a variant of mixture models that allow for possible memory in the sequence of hidden states. An HMM is thus defined as a pair of processes $(z_j,y_j, j=1,2,\ldots)$, where the latent sequence  $(z_j)$ is a  Markov chain and where the distribution of the observation $y_j$ at time $t_j$ depends on the state $z_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hmm0.png|caption=Dynamics of a hidden Markov model}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In a population approach, HMMs from several individuals can be described simultaneously by considering ''mixed'' HMMs.&lt;br /&gt;
Let $y_i=\left(y_{i,1},\ldots,y_{i,n_i}\right)$ and $z_i= \left(z_{i,1}, \ldots,z_{i,n_i}\right)$ denote respectively the sequences of observations and hidden states for individual $i$.&lt;br /&gt;
&lt;br /&gt;
We suppose that the joint distribution of $(z_i,y_i)$ is a parametric distribution that depends on a vector of parameters  $\psi_i$ and can be decomposed as&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:hmm1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyzipsii(z_i,y_i {{!}} \psi_i) = \pczipsii(z_i {{!}}\psi_i) \, \pcyizpsii(y_i {{!}} z_i,\psi_i) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
For each individual $i$, $z_i$ is a Markov chain whose  probability distribution is defined by&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the distribution $ \pi_{i,1} = (\pi_{i,1}^{k},\ k=1,2,\ldots,K)$ of the first state $z_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \pi_{i,1}^{k} = \prob{z_{i,1} = k {{!}} \psi_i} . &amp;lt;/math&amp;gt;  }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the sequence of ''transition matrices'' $(Q_{i,j} \ ; \, j=2,3,\ldots)$, where for each $j$, $Q_{i,j} = (q_{i,j}^{\ell,k} \ ; \, 1\leq \ell,k \leq K)$ is a matrix of size $K \times K$ such that $q_{i,j}^{\ell,k} = \prob{z_{i,j} = k | z_{i,j-1}=\ell , \psi_i}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=markov_1.png|caption=Transitions of a Markov chain with 3 states}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution $\qcyizpsii$  depends on the model for the observations: for each  state, observation $y_{ij}$ has a certain distribution. Let us see some examples:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Examples ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
1. In a continuous data model, one possibility is that  the residual error model is a hidden Markov model that can randomly switch between $K$ possible residual error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 1&lt;br /&gt;
|text=In this example, we consider a 2-state Markov chain.  A constant error model is assumed in each state:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; \sin(\alpha \, t_{ij}) + a_{i,1} \teps_{ij}   \quad \text{if } z_{ij}=1 \\&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; \sin(\alpha \, t_{ij}) + a_{i,2} \teps_{ij}   \quad \text{if } z_{ij}=2.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The figure below displays simulated data from this model for 4 individuals. Observations drawn from state 1 (resp.  state 2) are displayed in magenta (resp.  black). Of course, the states are unknown in the case of hidden Markov models, i.e., only the values are observed in  practice, not the colors.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:hmm1bis.png|link=]]&lt;br /&gt;
&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2. In a Poisson model for  count data, the Poisson parameter might randomly switch between $K$  intensities. Such models have been used for describing the evolution of seizures in epileptic patients:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 2&lt;br /&gt;
|text= Instead of assuming a single [http://en.wikipedia.org/wiki/Poisson_distribution Poisson distribution] for the observed numbers of seizures, this model assumes that patients go through alternating periods of low and high epileptic susceptibility.  Therefore we consider what is called a 2-state Poisson mixed-HMM:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;\sim&amp;amp; {\rm Poisson}(\lambda_{i,1})   \quad \text{if } z_{ij}=1 \\&lt;br /&gt;
y_{ij} &amp;amp;\sim&amp;amp; {\rm Poisson}(\lambda_{i,2})   \quad \text{if } z_{ij}=2.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
:: [[File:hmm2bis.png|link=]]&lt;br /&gt;
&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Distributions of  observations==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Assuming that the $N$ individuals are independent, the joint  pdf  is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:sdepdf&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcypsi(y_1,\ldots,y_N {{!}} \psi_1,\ldots,\psi_N ) = \prod_{i=1}^{N}\pcyipsii(y_i {{!}} \psi_i).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
Then, computing the conditional distribution of the observations $\qcyipsii$ for any individual $i$ requires  integration of the joint conditional distribution $\qcyzipsii$ over the states:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \sum_{z_i \in \mathbf{S} } \pcyzipsii(z_i, y_i {{!}} \psi_i) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \sum_{z_i \in \mathbf{S} } \pczipsii(z_i {{!}} \psi_i) \, \pcyizpsii(y_i {{!}} z_i,\psi_i) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \sum_{z_i \in \mathbf{S} } \left\{ \pi_{i,1}^{z_{i,1} } \pcyiONEzpsii(y_{i,1} {{!}} z_{i,1},\psi_i)\prod_{j=2}^{n} \left( q_{i,j}^{z_{i,j-1},z_{i,j} } \, \pcyijzpsii(y_{i,j} {{!}} z_{i,j},\psi_i) \right) \right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Though this looks complicated, it turns out that  forward recursion of the [http://en.wikipedia.org/wiki/Baum-Welch_algorithm Baum-Welch algorithm] provides a quick way to numerically compute it.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Albert1991,&lt;br /&gt;
title = &amp;quot;A two state Markov mixture model for a time series of epileptic seizure counts&amp;quot;,&lt;br /&gt;
author = &amp;quot;Albert, P. S.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Biometrics&amp;quot;,&lt;br /&gt;
volume = &amp;quot;47&amp;quot;,&lt;br /&gt;
year = &amp;quot;1991&amp;quot;,&lt;br /&gt;
pages = &amp;quot;1371-1381&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Altman2007,&lt;br /&gt;
title = &amp;quot;Mixed hidden Markov models : an extension of the hidden Markov model to the longitudinal data setting&amp;quot;,&lt;br /&gt;
author = &amp;quot;Altman, R. M.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Journal of the American Statistical Association&amp;quot;,&lt;br /&gt;
volume = &amp;quot;102&amp;quot;,&lt;br /&gt;
year = &amp;quot;2007&amp;quot;,&lt;br /&gt;
pages = &amp;quot;201-210&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Anisimov2007,&lt;br /&gt;
title = &amp;quot;Analysis of responses in migraine modelling using hidden Markov models&amp;quot;,&lt;br /&gt;
author = &amp;quot;Anisimov, W. and Maas, H. J. and Danhof, M. and Della Pasqua, O.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Statistics in Medicine&amp;quot;,&lt;br /&gt;
volume = &amp;quot;26&amp;quot;,&lt;br /&gt;
year = &amp;quot;2007&amp;quot;,&lt;br /&gt;
pages = &amp;quot;4163-4178&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{Cappe2005,&lt;br /&gt;
author = &amp;quot;Capp&amp;amp;eacute;e, O. and Moulines, E. and Ryd&amp;amp;eacute;en, T.&amp;quot;,&lt;br /&gt;
title = &amp;quot;Inference in hidden Markov models&amp;quot;,&lt;br /&gt;
year = &amp;quot;2005&amp;quot;,&lt;br /&gt;
publisher= &amp;quot;Springer Series in Statistics&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{ChaubertPereira2011,&lt;br /&gt;
title = &amp;quot;Markov and Semi-Markov Switching Linear Mixed Models Used to Identify&lt;br /&gt;
Forest Tree Growth Components&amp;quot;,&lt;br /&gt;
author = &amp;quot;Chaubert-Pereira, F. and Gu&amp;amp;eacute;don, Y. and Lavergne, C. and Trottier, C.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Biometrics&amp;quot;,&lt;br /&gt;
volume = &amp;quot;66&amp;quot;,&lt;br /&gt;
year = &amp;quot;2011&amp;quot;,&lt;br /&gt;
pages = &amp;quot;753-762&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delattre2012maximum,&lt;br /&gt;
  title={Maximum likelihood estimation in discrete mixed hidden Markov models using the SAEM algorithm},&lt;br /&gt;
  author={Delattre, M. and Lavielle, M.},&lt;br /&gt;
  journal={Computational Statistics &amp;amp; Data Analysis},&lt;br /&gt;
  year={2012},&lt;br /&gt;
  publisher={Elsevier}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delattre2012analysis,&lt;br /&gt;
  title={Analysis of exposure-response of CI-945 in patients with epilepsy: application of novel mixed hidden Markov modeling methodology},&lt;br /&gt;
  author={Delattre, M. and Savic, R. M. and Miller, R. and Karlsson, M. O. and Lavielle, M.},&lt;br /&gt;
  journal={Journal of pharmacokinetics and pharmacodynamics},&lt;br /&gt;
  pages={1-9},&lt;br /&gt;
  year={2012},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Maruotti2009,&lt;br /&gt;
title = &amp;quot;A semiparametric approach to hidden Markov models under longitudinal&lt;br /&gt;
observations&amp;quot;,&lt;br /&gt;
author = &amp;quot;Maruotti, A. and Ryd&amp;amp;eacute;en, T.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Statistics and Computing&amp;quot;,&lt;br /&gt;
volume = &amp;quot;19&amp;quot;,&lt;br /&gt;
year = &amp;quot;2009&amp;quot;,&lt;br /&gt;
pages = &amp;quot;381-393&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Rabiner1989,&lt;br /&gt;
title = &amp;quot;A tutorial on Hidden Markov Models and selected applications in speech recognition&amp;quot;,&lt;br /&gt;
author = &amp;quot;Rabiner, L. R.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Proceedings of the IEEE&amp;quot;,&lt;br /&gt;
volume = &amp;quot;77&amp;quot;,&lt;br /&gt;
year = &amp;quot;1989&amp;quot;,&lt;br /&gt;
pages = &amp;quot;257-286&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Rijmen2008,&lt;br /&gt;
title = &amp;quot;Qualitative longitudinal analysis of symptoms in patients with primary&lt;br /&gt;
and metastatic brain tumours&amp;quot;,&lt;br /&gt;
author = &amp;quot;Rijmen, F. and Ip, E. H. and Rapp, S. and Shaw, E. G.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Journal of the Royal Statistical Society - Series A.&amp;quot;,&lt;br /&gt;
volume = &amp;quot;171, Part 3&amp;quot;,&lt;br /&gt;
year = &amp;quot;2008&amp;quot;,&lt;br /&gt;
pages = &amp;quot;739-753&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack= Mixture models&lt;br /&gt;
|linkNext= Stochastic differential equations based models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Hidden_Markov_models&amp;diff=7439</id>
		<title>Hidden Markov models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Hidden_Markov_models&amp;diff=7439"/>
		<updated>2013-06-25T13:44:16Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Extensions chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Extensions]]&lt;br /&gt;
*[[Extensions| Introduction ]] | [[ Mixture models ]] | [[Hidden Markov models]]  | [[Stochastic differential equations based models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
[http://en.wikipedia.org/wiki/Markov_chain Markov chains] are a useful tool for analyzing categorical longitudinal data. However, sometimes the [https://en.wikipedia.org/wiki/Markov_process Markov process]  cannot be directly observed, though some output, dependent on the&lt;br /&gt;
(hidden) state, is visible. More precisely, we assume that the distribution of this  observable output depends on the underlying hidden state. Such models are called hidden Markov models (HMMs).&lt;br /&gt;
HMMs can be applied in many contexts and have turned out to be particularly pertinent in several biological contexts. For example, they are useful when characterizing diseases for which the existence of several discrete stages of illness is a realistic assumption, e.g., epilepsy and migraines.&lt;br /&gt;
&lt;br /&gt;
Here, we will consider a parametric framework with Markov chains in a discrete and finite state space $\mathbf{K} = \{1,\ldots,K\}$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Mixed hidden Markov models==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
HMMs have been developed to describe how a given system moves from one state to another over time, in situations where the successive visited states are unknown and a set of observations is the only available information to describe the dynamics of the system. HMMs can be seen as a variant of mixture models that allow for possible memory in the sequence of hidden states. An HMM is thus defined as a pair of processes $(z_j,y_j, j=1,2,\ldots)$, where the latent sequence  $(z_j)$ is a  [http://en.wikipedia.org/wiki/Markov_chain Markov chain] and where the distribution of the observation $y_j$ at time $t_j$ depends on the state $z_j$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hmm0.png|caption=Dynamics of a hidden Markov model}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In a population approach, HMMs from several individuals can be described simultaneously by considering ''mixed'' HMMs.&lt;br /&gt;
Let $y_i=\left(y_{i,1},\ldots,y_{i,n_i}\right)$ and $z_i= \left(z_{i,1}, \ldots,z_{i,n_i}\right)$ denote respectively the sequences of observations and hidden states for individual $i$.&lt;br /&gt;
&lt;br /&gt;
We suppose that the joint distribution of $(z_i,y_i)$ is a parametric distribution that depends on a vector of parameters  $\psi_i$ and can be decomposed as&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:hmm1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyzipsii(z_i,y_i {{!}} \psi_i) = \pczipsii(z_i {{!}}\psi_i) \, \pcyizpsii(y_i {{!}} z_i,\psi_i) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
For each individual $i$, $z_i$ is a Markov chain whose  probability distribution is defined by&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the distribution $ \pi_{i,1} = (\pi_{i,1}^{k},\ k=1,2,\ldots,K)$ of the first state $z_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \pi_{i,1}^{k} = \prob{z_{i,1} = k {{!}} \psi_i} . &amp;lt;/math&amp;gt;  }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the sequence of ''transition matrices'' $(Q_{i,j} \ ; \, j=2,3,\ldots)$, where for each $j$, $Q_{i,j} = (q_{i,j}^{\ell,k} \ ; \, 1\leq \ell,k \leq K)$ is a matrix of size $K \times K$ such that $q_{i,j}^{\ell,k} = \prob{z_{i,j} = k | z_{i,j-1}=\ell , \psi_i}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=markov_1.png|caption=Transitions of a Markov chain with 3 states}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution $\qcyizpsii$  depends on the model for the observations: for each  state, observation $y_{ij}$ has a certain distribution. Let us see some examples:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Examples ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
1. In a continuous data model, one possibility is that  the residual error model is a hidden Markov model that can randomly switch between $K$ possible residual error models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 1&lt;br /&gt;
|text=In this example, we consider a 2-state Markov chain.  A constant error model is assumed in each state:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; \sin(\alpha \, t_{ij}) + a_{i,1} \teps_{ij}   \quad \text{if } z_{ij}=1 \\&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; \sin(\alpha \, t_{ij}) + a_{i,2} \teps_{ij}   \quad \text{if } z_{ij}=2.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The figure below displays simulated data from this model for 4 individuals. Observations drawn from state 1 (resp.  state 2) are displayed in magenta (resp.  black). Of course, the states are unknown in the case of hidden Markov models, i.e., only the values are observed in  practice, not the colors.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:hmm1bis.png|link=]]&lt;br /&gt;
&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
2. In a Poisson model for  count data, the Poisson parameter might randomly switch between $K$  intensities. Such models have been used for describing the evolution of seizures in epileptic patients:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example 2&lt;br /&gt;
|text= Instead of assuming a single [http://en.wikipedia.org/wiki/Poisson_distribution Poisson distribution] for the observed numbers of seizures, this model assumes that patients go through alternating periods of low and high epileptic susceptibility.  Therefore we consider what is called a 2-state Poisson mixed-HMM:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;\sim&amp;amp; {\rm Poisson}(\lambda_{i,1})   \quad \text{if } z_{ij}=1 \\&lt;br /&gt;
y_{ij} &amp;amp;\sim&amp;amp; {\rm Poisson}(\lambda_{i,2})   \quad \text{if } z_{ij}=2.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
:: [[File:hmm2bis.png|link=]]&lt;br /&gt;
&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Distributions of  observations==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Assuming that the $N$ individuals are independent, the joint  pdf  is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:sdepdf&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcypsi(y_1,\ldots,y_N {{!}} \psi_1,\ldots,\psi_N ) = \prod_{i=1}^{N}\pcyipsii(y_i {{!}} \psi_i).&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
Then, computing the conditional distribution of the observations $\qcyipsii$ for any individual $i$ requires  integration of the joint conditional distribution $\qcyzipsii$ over the states:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \sum_{z_i \in \mathbf{S} } \pcyzipsii(z_i, y_i {{!}} \psi_i) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \sum_{z_i \in \mathbf{S} } \pczipsii(z_i {{!}} \psi_i) \, \pcyizpsii(y_i {{!}} z_i,\psi_i) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \sum_{z_i \in \mathbf{S} } \left\{ \pi_{i,1}^{z_{i,1} } \pcyiONEzpsii(y_{i,1} {{!}} z_{i,1},\psi_i)\prod_{j=2}^{n} \left( q_{i,j}^{z_{i,j-1},z_{i,j} } \, \pcyijzpsii(y_{i,j} {{!}} z_{i,j},\psi_i) \right) \right\} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Though this looks complicated, it turns out that  forward recursion of the [http://en.wikipedia.org/wiki/Baum-Welch_algorithm Baum-Welch algorithm] provides a quick way to numerically compute it.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Albert1991,&lt;br /&gt;
title = &amp;quot;A two state Markov mixture model for a time series of epileptic seizure counts&amp;quot;,&lt;br /&gt;
author = &amp;quot;Albert, P. S.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Biometrics&amp;quot;,&lt;br /&gt;
volume = &amp;quot;47&amp;quot;,&lt;br /&gt;
year = &amp;quot;1991&amp;quot;,&lt;br /&gt;
pages = &amp;quot;1371-1381&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Altman2007,&lt;br /&gt;
title = &amp;quot;Mixed hidden Markov models : an extension of the hidden Markov model to the longitudinal data setting&amp;quot;,&lt;br /&gt;
author = &amp;quot;Altman, R. M.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Journal of the American Statistical Association&amp;quot;,&lt;br /&gt;
volume = &amp;quot;102&amp;quot;,&lt;br /&gt;
year = &amp;quot;2007&amp;quot;,&lt;br /&gt;
pages = &amp;quot;201-210&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Anisimov2007,&lt;br /&gt;
title = &amp;quot;Analysis of responses in migraine modelling using hidden Markov models&amp;quot;,&lt;br /&gt;
author = &amp;quot;Anisimov, W. and Maas, H. J. and Danhof, M. and Della Pasqua, O.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Statistics in Medicine&amp;quot;,&lt;br /&gt;
volume = &amp;quot;26&amp;quot;,&lt;br /&gt;
year = &amp;quot;2007&amp;quot;,&lt;br /&gt;
pages = &amp;quot;4163-4178&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{Cappe2005,&lt;br /&gt;
author = &amp;quot;Capp&amp;amp;eacute;e, O. and Moulines, E. and Ryd&amp;amp;eacute;en, T.&amp;quot;,&lt;br /&gt;
title = &amp;quot;Inference in hidden Markov models&amp;quot;,&lt;br /&gt;
year = &amp;quot;2005&amp;quot;,&lt;br /&gt;
publisher= &amp;quot;Springer Series in Statistics&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{ChaubertPereira2011,&lt;br /&gt;
title = &amp;quot;Markov and Semi-Markov Switching Linear Mixed Models Used to Identify&lt;br /&gt;
Forest Tree Growth Components&amp;quot;,&lt;br /&gt;
author = &amp;quot;Chaubert-Pereira, F. and Gu&amp;amp;eacute;don, Y. and Lavergne, C. and Trottier, C.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Biometrics&amp;quot;,&lt;br /&gt;
volume = &amp;quot;66&amp;quot;,&lt;br /&gt;
year = &amp;quot;2011&amp;quot;,&lt;br /&gt;
pages = &amp;quot;753-762&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delattre2012maximum,&lt;br /&gt;
  title={Maximum likelihood estimation in discrete mixed hidden Markov models using the SAEM algorithm},&lt;br /&gt;
  author={Delattre, M. and Lavielle, M.},&lt;br /&gt;
  journal={Computational Statistics &amp;amp; Data Analysis},&lt;br /&gt;
  year={2012},&lt;br /&gt;
  publisher={Elsevier}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delattre2012analysis,&lt;br /&gt;
  title={Analysis of exposure-response of CI-945 in patients with epilepsy: application of novel mixed hidden Markov modeling methodology},&lt;br /&gt;
  author={Delattre, M. and Savic, R. M. and Miller, R. and Karlsson, M. O. and Lavielle, M.},&lt;br /&gt;
  journal={Journal of pharmacokinetics and pharmacodynamics},&lt;br /&gt;
  pages={1-9},&lt;br /&gt;
  year={2012},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Maruotti2009,&lt;br /&gt;
title = &amp;quot;A semiparametric approach to hidden Markov models under longitudinal&lt;br /&gt;
observations&amp;quot;,&lt;br /&gt;
author = &amp;quot;Maruotti, A. and Ryd&amp;amp;eacute;en, T.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Statistics and Computing&amp;quot;,&lt;br /&gt;
volume = &amp;quot;19&amp;quot;,&lt;br /&gt;
year = &amp;quot;2009&amp;quot;,&lt;br /&gt;
pages = &amp;quot;381-393&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Rabiner1989,&lt;br /&gt;
title = &amp;quot;A tutorial on Hidden Markov Models and selected applications in speech recognition&amp;quot;,&lt;br /&gt;
author = &amp;quot;Rabiner, L. R.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Proceedings of the IEEE&amp;quot;,&lt;br /&gt;
volume = &amp;quot;77&amp;quot;,&lt;br /&gt;
year = &amp;quot;1989&amp;quot;,&lt;br /&gt;
pages = &amp;quot;257-286&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{Rijmen2008,&lt;br /&gt;
title = &amp;quot;Qualitative longitudinal analysis of symptoms in patients with primary&lt;br /&gt;
and metastatic brain tumours&amp;quot;,&lt;br /&gt;
author = &amp;quot;Rijmen, F. and Ip, E. H. and Rapp, S. and Shaw, E. G.&amp;quot;,&lt;br /&gt;
journal = &amp;quot;Journal of the Royal Statistical Society - Series A.&amp;quot;,&lt;br /&gt;
volume = &amp;quot;171, Part 3&amp;quot;,&lt;br /&gt;
year = &amp;quot;2008&amp;quot;,&lt;br /&gt;
pages = &amp;quot;739-753&amp;quot;}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack= Mixture models&lt;br /&gt;
|linkNext= Stochastic differential equations based models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Mixture_models&amp;diff=7438</id>
		<title>Mixture models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Mixture_models&amp;diff=7438"/>
		<updated>2013-06-25T13:39:51Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Mixtures of mixed-effects models */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Extensions chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Extensions]]&lt;br /&gt;
*[[Extensions| Introduction ]] | [[ Mixture models ]] | [[Hidden Markov models]]  | [[Stochastic differential equations based models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
Mixed-effects models are frequently used for modeling longitudinal data when data is obtained from different individuals  from the same population. These models allow us  to take into account  between-subject variability.&lt;br /&gt;
One complicating factor arises when  data is obtained from a population with some underlying heterogeneity. If we assume that the population consists of several homogeneous sub-populations, a straightforward extension of  mixed-effects models is a finite mixture of mixed-effects models.&lt;br /&gt;
&lt;br /&gt;
As an example, the use of a mixture of mixed effects models is particularly relevant when the response of patients to a drug therapy is heterogeneous. In any clinical efficacy trial, patients who respond, partially respond or do not respond at all can be considered  different sub-populations with quite different profiles.&lt;br /&gt;
&lt;br /&gt;
The introduction of a categorical covariate (e.g., sex, [http://en.wikipedia.org/wiki/Genotype genotype], treatment, status, etc.) into such a model already supposes that the whole population can be decomposed into sub-populations. The covariate then serves as a ''label'' for assigning each individual to a sub-population. In practice, the covariate can either be known or not.&lt;br /&gt;
&lt;br /&gt;
Mixture models usually refer to  models for which the categorical covariate is unknown, but whatever the case, the joint model that brings together all the parts (observations, individual parameters, covariates, labels, design, etc.) is the same. The difference appears when having to perform certain tasks and in the methods needed to implement them. For instance, the task of simulation makes no distinction between the two situations because all the variables are simulated, whereas model construction is different depending on whether the labels are known or unknown: we have supervised learning if the labels are known and unsupervised learning otherwise.&lt;br /&gt;
&lt;br /&gt;
There exist several types of  mixture models which are useful in the context of mixed-effects models, e.g., mixtures of distributions, mixtures of residual error models, and mixtures of structural models.&lt;br /&gt;
Indeed, heterogeneity in the response variable cannot be always adequately explained only by inter-patient variability of certain parameters. It can  therefore be necessary to introduce diversity into the structural models themselves:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Between-subject model mixtures''  assume that there exist sub-populations of individuals. Here, various structural models describe the response of the different sub-populations, and each subject belongs to one sub-population. One can imagine for example different structural models for responders, non responders and partial responders to a given treatment.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* ''Within-subject model mixtures'' assume that there exist sub-populations (of cells, viruses, etc.) within each patient. Again, differing structural models describe the response of the different sub-populations, but the proportion of each sub-population depends on the patient.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Mixtures of mixed-effects models ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity, we will consider a basic model that involves individual parameters $\bpsi=(\psi_i,1\leq i \leq N)$ and observations $\by=(y_i,1\leq i \leq N)$, where $y_i=(y_{ij},1\leq j \leq n_i)$. Then, the simplest way to model a finite mixture model is to introduce a label sequence  $\bz=(z_i ; 1\leq z_i \leq N)$ that takes its values in $\{1,2,\ldots,M\}$ and is such that $z_i=m$ if subject $i$ belongs to sub-population $m$.&lt;br /&gt;
&lt;br /&gt;
In some situations, the label set $\bz$ is known and can then be used as a categorical covariate in the model.&lt;br /&gt;
If $\bz$ is known and if we consider $\bz$  the realization of a random vector,  the model is the conditional distribution&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;label{eq:mixt1}&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcypsiz(\by,\bpsi {{!}} \bz;\theta) = \pccypsiz(\by {{!}} \bpsi , \bz)\pcpsiz(\bpsi {{!}} \bz;\theta) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
If $\bz$ is unknown, it is modeled as a random vector and the model is the joint distribution&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;label{eq:mixt2}&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pypsiz(\by,\bpsi, \bz;\theta) = \pcypsiz(\by,\bpsi {{!}}\bz;\theta)\pz(\bz;\theta) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
We therefore consider that $\bz=(z_i)$ is a set  of independent random variables taking its values in $\{1,2,\ldots,M\}$: for $i=1,2,\ldots, N$, there exist $\pw_{i,1},\pw_{i,2},\ldots,\pw_{i,M}$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\prob{z_i = m} = \pw_{i,m} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
A simple model might assume that the $(z_i)$ are identically distributed: $\pw_{i,m} = \pw_{m}$ for $m=1,\ldots,M$.&lt;br /&gt;
But more complex models can be considered, assuming for instance that an individual's probabilities depend on its covariate values.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example&lt;br /&gt;
|text=&lt;br /&gt;
The [http://en.wikipedia.org/wiki/Hepatitis_C_virus Hepatitis C virus] (HCV) can be divided into six distinct genotypes. Genotype 1 is the most difficult to treat, whereas individuals with genotypes 2 and 3 are almost three times more likely to respond to the therapy of a combination of [http://en.wikipedia.org/wiki/Alpha_interferon alpha interferon] and [http://en.wikipedia.org/wiki/Ribavirin ribavirin]. &lt;br /&gt;
&lt;br /&gt;
Suppose we want to divide  patients infected with HCV into three outcome groups: patients who respond, partially respond or do not respond. It is valid to assume that an individual's probabilities for ending up in each of these groups depends on their value for the genotype covariate. }}&lt;br /&gt;
&lt;br /&gt;
In its most general form, a mixture of mixed-effects models assumes that there exist $M$ joint distributions $\pyipsii_{1}$, ..., $\pyipsii_{M}$ and  vectors of parameters $\theta_1$, ..., $\theta_M$ such that for any individual $i$, the  joint distribution of $y_i$ and $\psi_i$ becomes&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) = \sum_{m=1}^M \prob{z_i = m} \pyipsii_{m}(y_i,\psi_i;\theta_m) ,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\pypsi_{m}$ is the joint distribution of $(y_i,\psi_i)$ in group $m$ and where $\theta=(\theta_1,\ldots,\theta_M)$.&lt;br /&gt;
&lt;br /&gt;
The distribution  of the observations $y_i$  is therefore itself a mixture of $M$ distributions:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
\pyi(y_i;\theta) &amp;amp;=&amp;amp; \int \pyipsii(y_i,\psi_i;\theta) \, d \psi_i  \end{array}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:mixt3&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
&amp;amp; = &amp;amp; \sum_{m=1}^M \prob{z_i = m} \left( \int \pyipsii_{m}(y_i,\psi_i,\theta_m) \, d \psi_i \right)  &lt;br /&gt;
\end{array}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
&amp;amp; = &amp;amp; \sum_{m=1}^M \prob{z_i = m} \pyi_{m}(y_i;\theta_m) .   \end{array}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The mixture can then be looked at via the distribution of the individual parameters $\qpsii$ and/or the conditional distribution of the observations $\qcyipsii$.&lt;br /&gt;
Let us now see some examples of such mixtures models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A latency structure can  be introduced at the individual parameter level:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp; =&amp;amp; \pcyipsii(y_i {{!}} \psi_i)\ppsii(\psi_i;\theta) \\&lt;br /&gt;
&amp;amp; =&amp;amp;  \pcyipsii(y_i {{!}} \psi_i) \left(\sum_{m=1}^M \prob{z_i = m} \ppsii_{m}(\psi_i;\theta_m) \right) ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $\ppsii_{m}(\psi_i;\theta_m)$ is the  distribution of the individual parameters in group $m$. For example, a mixture of linear Gaussian  models  for the individual parameters assumes that there exist $M$ population parameters $\psi_{{\rm pop},1}, \ldots, \psi_{{\rm pop},M}$, vectors of coefficients $\beta_{1}, \ldots, \beta_{M}$,  variance matrices $\Omega_{1}, \ldots, \Omega_{M}$ and  transformations $h_1,\ldots,h_M$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
h_m(\psi_i) \ {{!}} \ z_i=m \ \ \sim \ \  {\cal N}(\mu_m , \Omega_m),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $\mu_m = h_m(\psi_{ {\rm pop},m})+ \langle \beta_m , c_i \rangle$.&lt;br /&gt;
&lt;br /&gt;
: This is the most general representation possible because it allows the transformation, population parameters, covariate model and variance-covariance structure of the random effects all to vary from one group to the next. A more simpler representation would have one or all of these  fixed across the groups.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* A latency structure can also be introduced at the level of the conditional distribution of the observations $(y_{ij})$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp; =&amp;amp; \pcyipsii(y_i {{!}} \psi_i)\ppsii(\psi_i;\theta) \\&lt;br /&gt;
&amp;amp; =&amp;amp;   \left(\sum_{m=1}^M \prob{z_i = m} \pcyipsii_{m}(y_i{{!}}\psi_i) \right) \ppsii(\psi_i;\theta) ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $\pcyipsii_{m}$ is the conditional distribution of the  observations in group $m$. For example, the  model for continuous data&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;mixturey&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
 y_{ij}  = f\left( t_{ij};\psi_i,z_i \right) + g\left( t_{ij};\psi_i,z_i \right)\teps_{ij}&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
: with $\teps_{ij} \sim {\cal N}(0,1)$, can be equivalently represented as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{ij}  {{!}} \,z_i=m \ \ \sim \ \ {\cal N}(f_m( t_{ij};\psi_i) , \ g_m( t_{ij};\psi_i)^2)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: for each $m=1,\ldots,M$. A mixture of conditional distributions  therefore reduces to a mixture of structural models and/or residual errors.&lt;br /&gt;
&lt;br /&gt;
: To give a precise example, a mixture of constant error models would assume that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij}  &amp;amp;= &amp;amp; f\left( t_{ij};\psi_i \right) + \left( \sum_{m=1}^M \one_{z_i = m} a_m \right) \varepsilon_{ij} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Alternatively, between subject model mixtures (BSMM) assume that the structural model is a mixture of $M$ different structural models:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;bsmm&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
f\left( t_{ij};\psi_i,z_i \right) = \sum_{m=1}^M \one_{z_i = m} f_m\left( t_{ij};\psi_i \right) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=It may be too simplistic to assume that each individual is represented by only one well-defined model from the mixture. For instance, in a pharmacological setting there may be sub-populations of cells or viruses ''within each patient'' that react differently to a drug treatment. In this case, it makes sense to consider that the mixture of models happens ''within'' each individual. Such within-subject model mixtures (WSMM) therefore require additional vectors of individual parameters $\pi_i=(\pi_{i,1},\ldots \pi_{i,M})$ representing proportions of the $M$ models within each individual $i$:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;wsmm&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
f\left( t_{ij};\psi_i,z_i \right) = \sum_{m=1}^M \pi_{i,m} f_m\left( t_{ij};\psi_i \right) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
The proportions $(\pi_{i,m})$ are now individual parameters in the model and the problem is transformed into a standard NLMEM.&lt;br /&gt;
These proportions are assumed to be positive and summing to $1$ for each patient. We can then define  $\pi_{i,m}$  in order to satisfy these constraints. One possible way to do this is:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\pi_{i,m} =\displaystyle{ \frac{\gamma_{i,m} }{\sum_{\ell=1}^M \gamma_{i,\ell} } }, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
 &lt;br /&gt;
where $\log(\gamma_{i,m}) \sim {\cal N}(\log(\gamma_{ {\rm pop},m}), \omega^2_m)$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Example 1: Mixtures of normal distributions==&lt;br /&gt;
&lt;br /&gt;
We consider here a simple PK model for a single oral administration:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
f(t ; ka,V,ke) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, the PK parameters are the  absorption rate constant $ka$, the elimination rate constant $ke$ and the volume of distribution $V$.&lt;br /&gt;
&lt;br /&gt;
We can model the PK parameters $\psi_i=(ka_i,V_i, ke_i)$ of individual $i$ randomly chosen from the population as a vector of independent random parameters.&lt;br /&gt;
&lt;br /&gt;
The figure shows the final distribution obtained for the volume when given as a mixture of two log-normal distributions:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log(V_i )  \sim 0.35 \ {\cal N}(\log(70) , 0.3^2) + 0.65 \ {\cal N}(\log(42) , 0.3^2). &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=pkmixt.png|caption= 2 log-normal distributions $p_1$ and $p_2$ for the volume and a mixture of these two distributions}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, the structural model $f$ is a function of time and  $f( t ; \psi_i)$ is the predicted concentration of the drug in individual $i$ at time $t$.&lt;br /&gt;
Then, $f( \, \cdot \, ; \psi_i)$ is a random function because it depends on a random parameter $\psi_i$.&lt;br /&gt;
The probability distribution of $f( \, \cdot \, ; \psi_i)$ is therefore that of the concentration predicted by the model. It represents the inter-individual variability of the drug's pharmacokinetics in the population.&lt;br /&gt;
&lt;br /&gt;
The figure below displays prediction  intervals for the  concentration $f( \, \cdot \, ; \psi_i)$ for one individual $i$ randomly chosen in the population, where $\psi_i=(ka_i,V_i,Cl_i)$ are the PK parameters. In other words, this plot allows us to visualize the impact of the inter-individual variability of the individual PK parameters on the exposure to the drug.&lt;br /&gt;
Here, $V_i$ is a mixture of two log-normal distributions as described above while $ka_i$ and $ke_i$ have log-normal distributions:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\log(ka_i ) &amp;amp;\sim&amp;amp; {\cal N}(\log(1) , 0.3^2) \\&lt;br /&gt;
\log(ke_i ) &amp;amp;\sim&amp;amp; {\cal N}(\log(4) , 0.3^2) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=pkmixture1.png|caption=Prediction intervals for the predicted concentration kinetics $f( \, \cdot \, ; \psi_i)$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
Here, the distribution of $f( \, \cdot \, ; \psi_i)$ cannot be computed in a closed form because the model is non-linear, but it can be easily estimated by Monte Carlo simulation.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, the distribution of $f( \, \cdot \, ; \psi_i)$  is  itself a mixture of 2 distributions since the distribution of $\psi_i$ is a mixture of distributions due to $V_i$. It is interesting to see the distribution of  the predicted concentration in each subpopulation. Indeed, any  individual $i$ will either have a log-volume from ${\cal N}(\log(70) , 0.3^2)$ (with probability 0.35) or a log-volume from ${\cal N}(\log(42) , 0.3^2)$ (with probability 0.65), so in order to visualize  what really happen to a single individual $i$, we need to split the data into two plots: 35% of the individuals  will have concentration kinetics distributed like on the left, and 65% like on the right.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=pkmixture2.png|caption=The probability distribution of the predicted concentration kinetics $f( \, \cdot \, ; \psi_i)$ in the two subpopulations}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Example 2: Mixtures of structural models==&lt;br /&gt;
&lt;br /&gt;
Here we are interested in a study which concerns  treated  HIV-infected patients.  The output data is the [http://en.wikipedia.org/wiki/Viral_load viral load] evolution for these  patients.&lt;br /&gt;
The figure below gives examples of patients with one of three &amp;quot;characteristic&amp;quot; viral load progressions:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Non-responders'' (1) show no decline in viral load.&lt;br /&gt;
&lt;br /&gt;
* ''Responders'' (2) exhibit a sustained viral load decline.&lt;br /&gt;
&lt;br /&gt;
* ''Rebounders'' (3 and 4) exhibit an initial drop in viral load, then a rebound to higher viral load levels.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hiv1.png|caption= Viral load progression for 4 HIV-infected patients. &amp;lt;br&amp;gt; (1) non-responder; (2) responder; (3) and (4) are rebounders. Red points indicate below level of quantification data.}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks:&lt;br /&gt;
|text= 	&amp;amp;#32;&lt;br /&gt;
* Since viral loads generally evolve exponentially over time, they are most commonly expressed on a logarithmic scale.&lt;br /&gt;
&lt;br /&gt;
*There is a detection limit at $50$ HIV RNA copies/ml, corresponding to a log-viral load of $1.7$, i.e., data are left-censored. These points are shown in red.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Within a few months of HIV infection, patients typically  enter a steady state of chronic infection and have a stabilized concentration of HIV-1 in [http://en.wikipedia.org/wiki/Blood_plasma blood plasma]. This concentration is modeled by an individual constant $A_{i,0}$. When  [http://en.wikipedia.org/wiki/Anti-retroviral anti-retroviral treatment] starts, the viral load of patients who respond shows an initial rapid [http://en.wikipedia.org/wiki/Exponential_decay exponential decay], usually followed by a slower second phase of exponential decay.&lt;br /&gt;
This two-phase decay in viral load can be approximated by the bi-exponential model:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;A_{1}e^{-\lambda_{1}t} +A_{2}e^{-\lambda_{2}t} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
After the decrease in viral load level, some subjects show a rebound, which can be due to several factors (non-adherence to the therapy, emergence of drug-resistant virus strains, etc.).&lt;br /&gt;
We propose to extend the bi-exponential model to these patients by adding a third phase, characterized by a logistic growth process $A_{3}/({1+e^{-\lambda_{3}(t-\tau)}})$, where $\tau$ is the inflection point of this growth process.&lt;br /&gt;
&lt;br /&gt;
We can then describe the log-transformed  viral load with a BSMM with  three simple models, corresponding to each of three characteristic viral load progressions:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
f_1(t_{ij},\psi_i) &amp;amp;=&amp;amp; A_{i,0}  \\&lt;br /&gt;
f_2(t_{ij},\psi_i) &amp;amp;=&amp;amp;A_{i,1}e^{-\lambda_{i,1}t_{ij} } +A_{i,2}e^{-\lambda_{i,2}t_{ij} } \\&lt;br /&gt;
f_3(t_{ij},\psi_i) &amp;amp;=&amp;amp;A_{i,1}e^{-\lambda_{i,1}t_{ij} } +A_{i,2}e^{-\lambda_{i,2}t_{ij} }&lt;br /&gt;
+ \displaystyle{ \frac{A_{i,3} }{1+e^{-\lambda_{i,3}(t_{ij}-\tau_i) } } } .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The log-transformed  viral load can  then be modeled  by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log(y_{ij} ) = \sum_{m=1}^3\one_{z_i=m}\log(f_m(t_{ij},\psi_i) ) + \varepsilon_{ij},&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $y_{ij}$ is the viral load for subject $i$ at time $t_{ij}$ and $\psi_i=(A_{i,0},A_{i,1},A_{i,2},A_{i,3},\lambda_{i,1},\lambda_{i,2},\lambda_{i,3},\tau_i)$ the vector of individual parameters.&lt;br /&gt;
&lt;br /&gt;
The figure below displays the predicted viral loads for the 4 patients using model $f_1$ for patient 1,  $f_2$ for patient 2 and  $f_3$ for patients 3 and 4, with the &amp;lt;balloon title=&amp;quot;these values were not obtained ''by chance'', they were estimated using Monolix, but that's another story...&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;	parameters&amp;lt;/balloon&amp;gt; given to the right of the figure:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;5&amp;quot; &lt;br /&gt;
|| &lt;br /&gt;
{{ImageWithCaption_special|image=hiv2.png|caption=Observed and predicted viral load progression for 4 HIV-infected patients}}&lt;br /&gt;
|| &lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;right&amp;quot; style=&amp;quot;width: 40%&amp;quot;&lt;br /&gt;
!| ID  || $A_0$  || $A_1$ || $A_2$ || $A_3$ || $\lambda_1$ || $\lambda_2$ || $\lambda_3$ || $\tau$&lt;br /&gt;
|-&lt;br /&gt;
|  1 || 92 || $-$ || $-$ || $-$ || $-$ || $-$ || $-$ || $-$&lt;br /&gt;
|-&lt;br /&gt;
|  2 || $-$ || 66 || 5 || $-$ || 0.14 || $2\times10^{-5}$ || $-$ || $-$&lt;br /&gt;
|-&lt;br /&gt;
| 3 || $-$ || 53 || 6 ||28 ||0.15 ||$1.5\times10^{-5}$ ||0.15 ||200&lt;br /&gt;
|-&lt;br /&gt;
| 4 || $-$  || 77 || 10 ||100 ||0.1 ||$1.5\times10^{-5}$ ||0.013 ||270 &lt;br /&gt;
|}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Not all observed viral load progressions fall so easily into one of the three classes, as for example the  patients shown in the next figure.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hiv3.png|caption= Viral load data for 4 patients with ambiguous progressions}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In these cases, it does not seem quite so reasonable to model the data under the BSMM assumption that each patient must belong uniquely to one class. Instead, it is perhaps more natural to suppose that each patient is partially responding, partially non-responding and partially rebounding  to the given drug treatment. The goal becomes to find the relative strength of each process in each patient, and  a WSMM is an ideal tool to do this. Without going further into the details, here are the resulting observed and predicted viral loads for these 4 individuals when each individual represents a mixture of the three viral load progressions.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hiv4.png|caption= Observed and predicted viral load progression for 4 patients time using WSMM}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{biernacki2006,&lt;br /&gt;
author = {Biernacki, C. and Celeux, G. and Govaert, G. and Langrognet, F.},&lt;br /&gt;
title  = {Model-Based Cluster and Discriminant Analysis with the MIXMOD Software. },&lt;br /&gt;
journal = {Computational Statistics and Data Analysis},&lt;br /&gt;
volume = {51},&lt;br /&gt;
number = {2},&lt;br /&gt;
pages = {587-600},&lt;br /&gt;
year = {2006}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{celeux2000a,&lt;br /&gt;
author = {Celeux, G.  and Hurn, M. and Robert, C.},&lt;br /&gt;
title  = {Computational and inferential difficulties with mixtures posterior distribution. },&lt;br /&gt;
journal = {J. American Statist. Assoc.},&lt;br /&gt;
volume = {95},&lt;br /&gt;
number = {3},&lt;br /&gt;
pages = {957-979},&lt;br /&gt;
year = {2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibitex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delacruz2008,&lt;br /&gt;
author = {De la Cruz, R. and Quintana, F. A. and Marshall, G.},&lt;br /&gt;
title  = {Model Based Clustering for Longitudinal Data},&lt;br /&gt;
journal = {Computational Statistics and Data Analysis},&lt;br /&gt;
volume = {52},&lt;br /&gt;
number = {3},&lt;br /&gt;
pages = {1441-1457},&lt;br /&gt;
year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fruhwirth2006,&lt;br /&gt;
author = {Fr&amp;amp;uuml;hwirth-Schnatter, S.},&lt;br /&gt;
title  = {Finite Mixture and Markov Switching Models},&lt;br /&gt;
publisher = {Springer},&lt;br /&gt;
pages = {},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
editor = {},&lt;br /&gt;
year = {2006}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hou2008,&lt;br /&gt;
author = {Hou, W. and Li, H. and Zhang, B. and Huang, M. and Wu, R.},&lt;br /&gt;
title  = {A nonlinear mixed-effect mixture model for functional mapping of dynamic traits},&lt;br /&gt;
journal = {Heredity},&lt;br /&gt;
volume = {101},&lt;br /&gt;
pages = {321-328},&lt;br /&gt;
year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{ketchum2012,&lt;br /&gt;
author = {Ketchum, J. M. and Best, A. M. and Ramakrishnan, V.},&lt;br /&gt;
title  = {A Within-Subject Normal-Mixture Model with Mixed-Effects for Analyzing Heart Rate Variability},&lt;br /&gt;
journal = {J. Biomet Biostat},&lt;br /&gt;
volume = {S7:013},&lt;br /&gt;
year = {2012}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{lavielle2013mixture,&lt;br /&gt;
  title={An improved SAEM algorithm for maximum likelihood estimation in mixtures of non linear mixed effects models},&lt;br /&gt;
  author={Lavielle, M. and Mbogning, C.},&lt;br /&gt;
  journal={Statistics &amp;amp; Computing  (to appear)},&lt;br /&gt;
  volume={},&lt;br /&gt;
  number={},&lt;br /&gt;
  pages={},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{mbogning2012between,&lt;br /&gt;
  title={Between-subject and within-subject model mixtures for classifying HIV treatment response},&lt;br /&gt;
  author={Mbogning, C. and Bleakley, K. and Lavielle, M.},&lt;br /&gt;
  journal={Progress in Applied Mathematics},&lt;br /&gt;
  volume={4},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={148-166},&lt;br /&gt;
  year={2012}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mclachland2000,&lt;br /&gt;
author = {McLachland, G. J. and Peel, D.},&lt;br /&gt;
title  = { Finite Mixture models. },&lt;br /&gt;
publisher = {Wiley-Interscience},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {2000},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{muthen1999finite,&lt;br /&gt;
  title={Finite mixture modeling with mixture outcomes using the EM algorithm},&lt;br /&gt;
  author={Muth&amp;amp;eacute;n, B. and Shedden, K.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={55},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={463-469},&lt;br /&gt;
  year={1999},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{ng2006,&lt;br /&gt;
author = {Ng, S. K. and McLachlan, G. J. and Wang, K. and Ben-Tovim, L. and Ng, S. W.},&lt;br /&gt;
title  = {A mixture model with mixed effects components for clustering correlated gene-expression profiles},&lt;br /&gt;
journal = {Bioinformatics},&lt;br /&gt;
volume = {22},&lt;br /&gt;
pages = {1745-1752},&lt;br /&gt;
year = {2006}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{rosner1997,&lt;br /&gt;
author = {Rosner, G. L. and Muller, P.},&lt;br /&gt;
title  = {Bayesian population pharmacokinetic and pharmacodynamic analyses using mixture models.},&lt;br /&gt;
journal = {J. Pharmacokin. Biopharm.},&lt;br /&gt;
volume = {25},&lt;br /&gt;
pages = {209-233},&lt;br /&gt;
year = {1997}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{verbeke1996linear,&lt;br /&gt;
  title={A linear mixed-effects model with heterogeneity in the random-effects population},&lt;br /&gt;
  author={Verbeke, G. and Lesaffre, E.},&lt;br /&gt;
  journal={Journal of the American Statistical Association},&lt;br /&gt;
  volume={91},&lt;br /&gt;
  number={433},&lt;br /&gt;
  pages={217-221},&lt;br /&gt;
  year={1996},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis Group}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wang2007,&lt;br /&gt;
author = {Wang, X. and Schumitzky, A. and D'Argenio, D. Z.},&lt;br /&gt;
title = {Non linear random effects mixture models : Maximum likelihood estimation via the EM algorithm },&lt;br /&gt;
journal = {Comput. Stat. Data Anal.},&lt;br /&gt;
volume = {51},&lt;br /&gt;
pages = {6614-6623},&lt;br /&gt;
year = {2007}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Extensions &lt;br /&gt;
|linkNext=Hidden Markov models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Mixture_models&amp;diff=7437</id>
		<title>Mixture models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Mixture_models&amp;diff=7437"/>
		<updated>2013-06-25T13:30:36Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Extensions chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Extensions]]&lt;br /&gt;
*[[Extensions| Introduction ]] | [[ Mixture models ]] | [[Hidden Markov models]]  | [[Stochastic differential equations based models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
Mixed-effects models are frequently used for modeling longitudinal data when data is obtained from different individuals  from the same population. These models allow us  to take into account  between-subject variability.&lt;br /&gt;
One complicating factor arises when  data is obtained from a population with some underlying heterogeneity. If we assume that the population consists of several homogeneous sub-populations, a straightforward extension of  mixed-effects models is a finite mixture of mixed-effects models.&lt;br /&gt;
&lt;br /&gt;
As an example, the use of a mixture of mixed effects models is particularly relevant when the response of patients to a drug therapy is heterogeneous. In any clinical efficacy trial, patients who respond, partially respond or do not respond at all can be considered  different sub-populations with quite different profiles.&lt;br /&gt;
&lt;br /&gt;
The introduction of a categorical covariate (e.g., sex, [http://en.wikipedia.org/wiki/Genotype genotype], treatment, status, etc.) into such a model already supposes that the whole population can be decomposed into sub-populations. The covariate then serves as a ''label'' for assigning each individual to a sub-population. In practice, the covariate can either be known or not.&lt;br /&gt;
&lt;br /&gt;
Mixture models usually refer to  models for which the categorical covariate is unknown, but whatever the case, the joint model that brings together all the parts (observations, individual parameters, covariates, labels, design, etc.) is the same. The difference appears when having to perform certain tasks and in the methods needed to implement them. For instance, the task of simulation makes no distinction between the two situations because all the variables are simulated, whereas model construction is different depending on whether the labels are known or unknown: we have supervised learning if the labels are known and unsupervised learning otherwise.&lt;br /&gt;
&lt;br /&gt;
There exist several types of  mixture models which are useful in the context of mixed-effects models, e.g., mixtures of distributions, mixtures of residual error models, and mixtures of structural models.&lt;br /&gt;
Indeed, heterogeneity in the response variable cannot be always adequately explained only by inter-patient variability of certain parameters. It can  therefore be necessary to introduce diversity into the structural models themselves:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Between-subject model mixtures''  assume that there exist sub-populations of individuals. Here, various structural models describe the response of the different sub-populations, and each subject belongs to one sub-population. One can imagine for example different structural models for responders, non responders and partial responders to a given treatment.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* ''Within-subject model mixtures'' assume that there exist sub-populations (of cells, viruses, etc.) within each patient. Again, differing structural models describe the response of the different sub-populations, but the proportion of each sub-population depends on the patient.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Mixtures of mixed-effects models ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity, we will consider a basic model that involves individual parameters $\bpsi=(\psi_i,1\leq i \leq N)$ and observations $\by=(y_i,1\leq i \leq N)$, where $y_i=(y_{ij},1\leq j \leq n_i)$. Then, the simplest way to model a finite mixture model is to introduce a label sequence  $\bz=(z_i ; 1\leq z_i \leq N)$ that takes its values in $\{1,2,\ldots,M\}$ and is such that $z_i=m$ if subject $i$ belongs to sub-population $m$.&lt;br /&gt;
&lt;br /&gt;
In some situations, the label set $\bz$ is known and can then be used as a categorical covariate in the model.&lt;br /&gt;
If $\bz$ is known and if we consider $\bz$  the realization of a random vector,  the model is the conditional distribution&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;label{eq:mixt1}&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcypsiz(\by,\bpsi {{!}} \bz;\theta) = \pccypsiz(\by {{!}} \bpsi , \bz)\pcpsiz(\bpsi {{!}} \bz;\theta) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
If $\bz$ is unknown, it is modeled as a random vector and the model is the joint distribution&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;label{eq:mixt2}&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pypsiz(\by,\bpsi, \bz;\theta) = \pcypsiz(\by,\bpsi {{!}}\bz;\theta)\pz(\bz;\theta) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
We therefore consider that $\bz=(z_i)$ is a set  of independent random variables taking its values in $\{1,2,\ldots,M\}$: for $i=1,2,\ldots, N$, there exist $\pw_{i,1},\pw_{i,2},\ldots,\pw_{i,M}$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\prob{z_i = m} = \pw_{i,m} . &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
A simple model might assume that the $(z_i)$ are identically distributed: $\pw_{i,m} = \pw_{m}$ for $m=1,\ldots,M$.&lt;br /&gt;
But more complex models can be considered, assuming for instance that an individual's probabilities depend on its covariate values.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example&lt;br /&gt;
|text=&lt;br /&gt;
The Hepatitis C virus (HCV) can be divided into six distinct genotypes. Genotype 1 is the most difficult to treat, whereas individuals with genotypes 2 and 3 are almost three times more likely to respond to the therapy of a combination of alpha interferon and ribavirin. &lt;br /&gt;
&lt;br /&gt;
Suppose we want to divide  patients infected with HCV into three outcome groups: patients who respond, partially respond or do not respond. It is valid to assume that an individual's probabilities for ending up in each of these groups depends on their value for the genotype covariate. }}&lt;br /&gt;
&lt;br /&gt;
In its most general form, a mixture of mixed-effects models assumes that there exist $M$ joint distributions $\pyipsii_{1}$, ..., $\pyipsii_{M}$ and  vectors of parameters $\theta_1$, ..., $\theta_M$ such that for any individual $i$, the  joint distribution of $y_i$ and $\psi_i$ becomes&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) = \sum_{m=1}^M \prob{z_i = m} \pyipsii_{m}(y_i,\psi_i;\theta_m) ,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $\pypsi_{m}$ is the joint distribution of $(y_i,\psi_i)$ in group $m$ and where $\theta=(\theta_1,\ldots,\theta_M)$.&lt;br /&gt;
&lt;br /&gt;
The distribution  of the observations $y_i$  is therefore itself a mixture of $M$ distributions:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
\pyi(y_i;\theta) &amp;amp;=&amp;amp; \int \pyipsii(y_i,\psi_i;\theta) \, d \psi_i  \end{array}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:mixt3&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
&amp;amp; = &amp;amp; \sum_{m=1}^M \prob{z_i = m} \left( \int \pyipsii_{m}(y_i,\psi_i,\theta_m) \, d \psi_i \right)  &lt;br /&gt;
\end{array}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
&amp;amp; = &amp;amp; \sum_{m=1}^M \prob{z_i = m} \pyi_{m}(y_i;\theta_m) .   \end{array}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The mixture can then be looked at via the distribution of the individual parameters $\qpsii$ and/or the conditional distribution of the observations $\qcyipsii$.&lt;br /&gt;
Let us now see some examples of such mixtures models.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* A latency structure can  be introduced at the individual parameter level:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp; =&amp;amp; \pcyipsii(y_i {{!}} \psi_i)\ppsii(\psi_i;\theta) \\&lt;br /&gt;
&amp;amp; =&amp;amp;  \pcyipsii(y_i {{!}} \psi_i) \left(\sum_{m=1}^M \prob{z_i = m} \ppsii_{m}(\psi_i;\theta_m) \right) ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $\ppsii_{m}(\psi_i;\theta_m)$ is the  distribution of the individual parameters in group $m$. For example, a mixture of linear Gaussian  models  for the individual parameters assumes that there exist $M$ population parameters $\psi_{{\rm pop},1}, \ldots, \psi_{{\rm pop},M}$, vectors of coefficients $\beta_{1}, \ldots, \beta_{M}$,  variance matrices $\Omega_{1}, \ldots, \Omega_{M}$ and  transformations $h_1,\ldots,h_M$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
h_m(\psi_i) \ {{!}} \ z_i=m \ \ \sim \ \  {\cal N}(\mu_m , \Omega_m),&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $\mu_m = h_m(\psi_{ {\rm pop},m})+ \langle \beta_m , c_i \rangle$.&lt;br /&gt;
&lt;br /&gt;
: This is the most general representation possible because it allows the transformation, population parameters, covariate model and variance-covariance structure of the random effects all to vary from one group to the next. A more simpler representation would have one or all of these  fixed across the groups.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* A latency structure can also be introduced at the level of the conditional distribution of the observations $(y_{ij})$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp; =&amp;amp; \pcyipsii(y_i {{!}} \psi_i)\ppsii(\psi_i;\theta) \\&lt;br /&gt;
&amp;amp; =&amp;amp;   \left(\sum_{m=1}^M \prob{z_i = m} \pcyipsii_{m}(y_i{{!}}\psi_i) \right) \ppsii(\psi_i;\theta) ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $\pcyipsii_{m}$ is the conditional distribution of the  observations in group $m$. For example, the  model for continuous data&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;mixturey&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
 y_{ij}  = f\left( t_{ij};\psi_i,z_i \right) + g\left( t_{ij};\psi_i,z_i \right)\teps_{ij}&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(4) }}&lt;br /&gt;
&lt;br /&gt;
: with $\teps_{ij} \sim {\cal N}(0,1)$, can be equivalently represented as&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_{ij}  {{!}} \,z_i=m \ \ \sim \ \ {\cal N}(f_m( t_{ij};\psi_i) , \ g_m( t_{ij};\psi_i)^2)&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: for each $m=1,\ldots,M$. A mixture of conditional distributions  therefore reduces to a mixture of structural models and/or residual errors.&lt;br /&gt;
&lt;br /&gt;
: To give a precise example, a mixture of constant error models would assume that&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij}  &amp;amp;= &amp;amp; f\left( t_{ij};\psi_i \right) + \left( \sum_{m=1}^M \one_{z_i = m} a_m \right) \varepsilon_{ij} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Alternatively, between subject model mixtures (BSMM) assume that the structural model is a mixture of $M$ different structural models:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;bsmm&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
f\left( t_{ij};\psi_i,z_i \right) = \sum_{m=1}^M \one_{z_i = m} f_m\left( t_{ij};\psi_i \right) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(5) }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=It may be too simplistic to assume that each individual is represented by only one well-defined model from the mixture. For instance, in a pharmacological setting there may be subpopulations of cells or viruses ''within each patient'' that react differently to a drug treatment. In this case, it makes sense to consider that the mixture of models happens ''within'' each individual. Such within-subject model mixtures (WSMM) therefore require additional vectors of individual parameters $\pi_i=(\pi_{i,1},\ldots \pi_{i,M})$ representing proportions of the $M$ models within each individual $i$:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;wsmm&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
f\left( t_{ij};\psi_i,z_i \right) = \sum_{m=1}^M \pi_{i,m} f_m\left( t_{ij};\psi_i \right) .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(6) }}&lt;br /&gt;
&lt;br /&gt;
The proportions $(\pi_{i,m})$ are now individual parameters in the model and the problem is transformed into a standard NLMEM.&lt;br /&gt;
These proportions are assumed to be positive and summing to $1$ for each patient. We can then define  $\pi_{i,m}$  in order to satisfy these constraints. One possible way to do this is:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\pi_{i,m} =\displaystyle{ \frac{\gamma_{i,m} }{\sum_{\ell=1}^M \gamma_{i,\ell} } }, &amp;lt;/math&amp;gt; }}&lt;br /&gt;
 &lt;br /&gt;
where $\log(\gamma_{i,m}) \sim {\cal N}(\log(\gamma_{ {\rm pop},m}), \omega^2_m)$.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Example 1: Mixtures of normal distributions==&lt;br /&gt;
&lt;br /&gt;
We consider here a simple PK model for a single oral administration:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
f(t ; ka,V,ke) &amp;amp;=&amp;amp; \frac{D\, k_a}{V(k_a-k_e)} \left( e^{-k_e \, t} - e^{-k_a \, t} \right).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, the PK parameters are the  absorption rate constant $ka$, the elimination rate constant $ke$ and the volume of distribution $V$.&lt;br /&gt;
&lt;br /&gt;
We can model the PK parameters $\psi_i=(ka_i,V_i, ke_i)$ of individual $i$ randomly chosen from the population as a vector of independent random parameters.&lt;br /&gt;
&lt;br /&gt;
The figure shows the final distribution obtained for the volume when given as a mixture of two log-normal distributions:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log(V_i )  \sim 0.35 \ {\cal N}(\log(70) , 0.3^2) + 0.65 \ {\cal N}(\log(42) , 0.3^2). &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=pkmixt.png|caption= 2 log-normal distributions $p_1$ and $p_2$ for the volume and a mixture of these two distributions}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, the structural model $f$ is a function of time and  $f( t ; \psi_i)$ is the predicted concentration of the drug in individual $i$ at time $t$.&lt;br /&gt;
Then, $f( \, \cdot \, ; \psi_i)$ is a random function because it depends on a random parameter $\psi_i$.&lt;br /&gt;
The probability distribution of $f( \, \cdot \, ; \psi_i)$ is therefore that of the concentration predicted by the model. It represents the inter-individual variability of the drug's pharmacokinetics in the population.&lt;br /&gt;
&lt;br /&gt;
The figure below displays prediction  intervals for the  concentration $f( \, \cdot \, ; \psi_i)$ for one individual $i$ randomly chosen in the population, where $\psi_i=(ka_i,V_i,Cl_i)$ are the PK parameters. In other words, this plot allows us to visualize the impact of the inter-individual variability of the individual PK parameters on the exposure to the drug.&lt;br /&gt;
Here, $V_i$ is a mixture of two log-normal distributions as described above while $ka_i$ and $ke_i$ have log-normal distributions:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\log(ka_i ) &amp;amp;\sim&amp;amp; {\cal N}(\log(1) , 0.3^2) \\&lt;br /&gt;
\log(ke_i ) &amp;amp;\sim&amp;amp; {\cal N}(\log(4) , 0.3^2) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=pkmixture1.png|caption=Prediction intervals for the predicted concentration kinetics $f( \, \cdot \, ; \psi_i)$}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text=&lt;br /&gt;
Here, the distribution of $f( \, \cdot \, ; \psi_i)$ cannot be computed in a closed form because the model is non-linear, but it can be easily estimated by Monte Carlo simulation.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, the distribution of $f( \, \cdot \, ; \psi_i)$  is  itself a mixture of 2 distributions since the distribution of $\psi_i$ is a mixture of distributions due to $V_i$. It is interesting to see the distribution of  the predicted concentration in each subpopulation. Indeed, any  individual $i$ will either have a log-volume from ${\cal N}(\log(70) , 0.3^2)$ (with probability 0.35) or a log-volume from ${\cal N}(\log(42) , 0.3^2)$ (with probability 0.65), so in order to visualize  what really happen to a single individual $i$, we need to split the data into two plots: 35% of the individuals  will have concentration kinetics distributed like on the left, and 65% like on the right.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=pkmixture2.png|caption=The probability distribution of the predicted concentration kinetics $f( \, \cdot \, ; \psi_i)$ in the two subpopulations}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Example 2: Mixtures of structural models==&lt;br /&gt;
&lt;br /&gt;
Here we are interested in a study which concerns  treated  HIV-infected patients.  The output data is the viral load evolution for these  patients.&lt;br /&gt;
The figure below gives examples of patients with one of three &amp;quot;characteristic&amp;quot; viral load progressions:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Non-responders'' (1) show no decline in viral load.&lt;br /&gt;
&lt;br /&gt;
* ''Responders'' (2) exhibit a sustained viral load decline.&lt;br /&gt;
&lt;br /&gt;
* ''Rebounders'' (3 and 4) exhibit an initial drop in viral load, then a rebound to higher viral load levels.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hiv1.png|caption= Viral load progression for 4 HIV-infected patients. &amp;lt;br&amp;gt; (1) non-responder; (2) responder; (3) and (4) are rebounders. Red points indicate below level of quantification data.}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks:&lt;br /&gt;
|text= 	&amp;amp;#32;&lt;br /&gt;
* Since viral loads generally evolve exponentially over time, they are most commonly expressed on a logarithmic scale.&lt;br /&gt;
&lt;br /&gt;
*There is a detection limit at $50$ HIV RNA copies/ml, corresponding to a log-viral load of $1.7$, i.e., data are left-censored. These points are shown in red.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Within a few months of HIV infection, patients typically  enter a steady state of chronic infection and have a stabilized concentration of HIV-1 in blood plasma. This concentration is modeled by an individual constant $A_{i,0}$. When  anti-retroviral treatment starts,   the viral load of patients who respond shows an initial rapid exponential decay, usually followed by a slower second phase of exponential decay.&lt;br /&gt;
This two-phase decay in viral load can be approximated by the bi-exponential model:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;A_{1}e^{-\lambda_{1}t} +A_{2}e^{-\lambda_{2}t} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
After the decrease in viral load level, some subjects show a rebound, which can be due to several factors (non-adherence to the therapy, emergence of drug-resistant virus strains, etc.).&lt;br /&gt;
We propose to extend the bi-exponential model to these patients by adding a third phase, characterized by a logistic growth process $A_{3}/({1+e^{-\lambda_{3}(t-\tau)}})$, where $\tau$ is the inflection point of this growth process.&lt;br /&gt;
&lt;br /&gt;
We can then describe the log-transformed  viral load with a BSMM with  three simple models, corresponding to each of three characteristic viral load progressions:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
f_1(t_{ij},\psi_i) &amp;amp;=&amp;amp; A_{i,0}  \\&lt;br /&gt;
f_2(t_{ij},\psi_i) &amp;amp;=&amp;amp;A_{i,1}e^{-\lambda_{i,1}t_{ij} } +A_{i,2}e^{-\lambda_{i,2}t_{ij} } \\&lt;br /&gt;
f_3(t_{ij},\psi_i) &amp;amp;=&amp;amp;A_{i,1}e^{-\lambda_{i,1}t_{ij} } +A_{i,2}e^{-\lambda_{i,2}t_{ij} }&lt;br /&gt;
+ \displaystyle{ \frac{A_{i,3} }{1+e^{-\lambda_{i,3}(t_{ij}-\tau_i) } } } .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The log-transformed  viral load can  then be modeled  by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\log(y_{ij} ) = \sum_{m=1}^3\one_{z_i=m}\log(f_m(t_{ij},\psi_i) ) + \varepsilon_{ij},&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $y_{ij}$ is the viral load for subject $i$ at time $t_{ij}$ and $\psi_i=(A_{i,0},A_{i,1},A_{i,2},A_{i,3},\lambda_{i,1},\lambda_{i,2},\lambda_{i,3},\tau_i)$ the vector of individual parameters.&lt;br /&gt;
&lt;br /&gt;
The figure below displays the predicted viral loads for the 4 patients using model $f_1$ for patient 1,  $f_2$ for patient 2 and  $f_3$ for patients 3 and 4, with the &amp;lt;balloon title=&amp;quot;these values were not obtained ''by chance'', they were estimated using Monolix, but that's another story...&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;	parameters&amp;lt;/balloon&amp;gt; given to the right of the figure:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{| cellpadding=&amp;quot;5&amp;quot; cellspacing=&amp;quot;5&amp;quot; &lt;br /&gt;
|| &lt;br /&gt;
{{ImageWithCaption_special|image=hiv2.png|caption=Observed and predicted viral load progression for 4 HIV-infected patients}}&lt;br /&gt;
|| &lt;br /&gt;
{| class=&amp;quot;wikitable&amp;quot; align=&amp;quot;right&amp;quot; style=&amp;quot;width: 40%&amp;quot;&lt;br /&gt;
!| ID  || $A_0$  || $A_1$ || $A_2$ || $A_3$ || $\lambda_1$ || $\lambda_2$ || $\lambda_3$ || $\tau$&lt;br /&gt;
|-&lt;br /&gt;
|  1 || 92 || $-$ || $-$ || $-$ || $-$ || $-$ || $-$ || $-$&lt;br /&gt;
|-&lt;br /&gt;
|  2 || $-$ || 66 || 5 || $-$ || 0.14 || $2\times10^{-5}$ || $-$ || $-$&lt;br /&gt;
|-&lt;br /&gt;
| 3 || $-$ || 53 || 6 ||28 ||0.15 ||$1.5\times10^{-5}$ ||0.15 ||200&lt;br /&gt;
|-&lt;br /&gt;
| 4 || $-$  || 77 || 10 ||100 ||0.1 ||$1.5\times10^{-5}$ ||0.013 ||270 &lt;br /&gt;
|}&lt;br /&gt;
|}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Not all observed viral load progressions fall so easily into one of the three classes, as for example the  patients shown in the next figure.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hiv3.png|caption= Viral load data for 4 patients with ambiguous progressions}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In these cases, it does not seem quite so reasonable to model the data under the BSMM assumption that each patient must belong uniquely to one class. Instead, it is perhaps more natural to suppose that each patient is partially responding, partially non-responding and partially rebounding  to the given drug treatment. The goal becomes to find the relative strength of each process in&lt;br /&gt;
each patient, and  a WSMM is an ideal tool to do this.&lt;br /&gt;
Without going further into the details, here are the resulting observed and predicted viral loads for these 4 individuals when each individual represents a mixture of the three viral load progressions.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=hiv4.png|caption= Observed and predicted viral load progression for 4 patients time using WSMM}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{biernacki2006,&lt;br /&gt;
author = {Biernacki, C. and Celeux, G. and Govaert, G. and Langrognet, F.},&lt;br /&gt;
title  = {Model-Based Cluster and Discriminant Analysis with the MIXMOD Software. },&lt;br /&gt;
journal = {Computational Statistics and Data Analysis},&lt;br /&gt;
volume = {51},&lt;br /&gt;
number = {2},&lt;br /&gt;
pages = {587-600},&lt;br /&gt;
year = {2006}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{celeux2000a,&lt;br /&gt;
author = {Celeux, G.  and Hurn, M. and Robert, C.},&lt;br /&gt;
title  = {Computational and inferential difficulties with mixtures posterior distribution. },&lt;br /&gt;
journal = {J. American Statist. Assoc.},&lt;br /&gt;
volume = {95},&lt;br /&gt;
number = {3},&lt;br /&gt;
pages = {957-979},&lt;br /&gt;
year = {2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibitex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{delacruz2008,&lt;br /&gt;
author = {De la Cruz, R. and Quintana, F. A. and Marshall, G.},&lt;br /&gt;
title  = {Model Based Clustering for Longitudinal Data},&lt;br /&gt;
journal = {Computational Statistics and Data Analysis},&lt;br /&gt;
volume = {52},&lt;br /&gt;
number = {3},&lt;br /&gt;
pages = {1441-1457},&lt;br /&gt;
year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fruhwirth2006,&lt;br /&gt;
author = {Fr&amp;amp;uuml;hwirth-Schnatter, S.},&lt;br /&gt;
title  = {Finite Mixture and Markov Switching Models},&lt;br /&gt;
publisher = {Springer},&lt;br /&gt;
pages = {},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
editor = {},&lt;br /&gt;
year = {2006}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hou2008,&lt;br /&gt;
author = {Hou, W. and Li, H. and Zhang, B. and Huang, M. and Wu, R.},&lt;br /&gt;
title  = {A nonlinear mixed-effect mixture model for functional mapping of dynamic traits},&lt;br /&gt;
journal = {Heredity},&lt;br /&gt;
volume = {101},&lt;br /&gt;
pages = {321-328},&lt;br /&gt;
year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{ketchum2012,&lt;br /&gt;
author = {Ketchum, J. M. and Best, A. M. and Ramakrishnan, V.},&lt;br /&gt;
title  = {A Within-Subject Normal-Mixture Model with Mixed-Effects for Analyzing Heart Rate Variability},&lt;br /&gt;
journal = {J. Biomet Biostat},&lt;br /&gt;
volume = {S7:013},&lt;br /&gt;
year = {2012}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{lavielle2013mixture,&lt;br /&gt;
  title={An improved SAEM algorithm for maximum likelihood estimation in mixtures of non linear mixed effects models},&lt;br /&gt;
  author={Lavielle, M. and Mbogning, C.},&lt;br /&gt;
  journal={Statistics &amp;amp; Computing  (to appear)},&lt;br /&gt;
  volume={},&lt;br /&gt;
  number={},&lt;br /&gt;
  pages={},&lt;br /&gt;
  year={2013}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{mbogning2012between,&lt;br /&gt;
  title={Between-subject and within-subject model mixtures for classifying HIV treatment response},&lt;br /&gt;
  author={Mbogning, C. and Bleakley, K. and Lavielle, M.},&lt;br /&gt;
  journal={Progress in Applied Mathematics},&lt;br /&gt;
  volume={4},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={148-166},&lt;br /&gt;
  year={2012}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mclachland2000,&lt;br /&gt;
author = {McLachland, G. J. and Peel, D.},&lt;br /&gt;
title  = { Finite Mixture models. },&lt;br /&gt;
publisher = {Wiley-Interscience},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {2000},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{muthen1999finite,&lt;br /&gt;
  title={Finite mixture modeling with mixture outcomes using the EM algorithm},&lt;br /&gt;
  author={Muth&amp;amp;eacute;n, B. and Shedden, K.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={55},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={463-469},&lt;br /&gt;
  year={1999},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{ng2006,&lt;br /&gt;
author = {Ng, S. K. and McLachlan, G. J. and Wang, K. and Ben-Tovim, L. and Ng, S. W.},&lt;br /&gt;
title  = {A mixture model with mixed effects components for clustering correlated gene-expression profiles},&lt;br /&gt;
journal = {Bioinformatics},&lt;br /&gt;
volume = {22},&lt;br /&gt;
pages = {1745-1752},&lt;br /&gt;
year = {2006}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{rosner1997,&lt;br /&gt;
author = {Rosner, G. L. and Muller, P.},&lt;br /&gt;
title  = {Bayesian population pharmacokinetic and pharmacodynamic analyses using mixture models.},&lt;br /&gt;
journal = {J. Pharmacokin. Biopharm.},&lt;br /&gt;
volume = {25},&lt;br /&gt;
pages = {209-233},&lt;br /&gt;
year = {1997}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{verbeke1996linear,&lt;br /&gt;
  title={A linear mixed-effects model with heterogeneity in the random-effects population},&lt;br /&gt;
  author={Verbeke, G. and Lesaffre, E.},&lt;br /&gt;
  journal={Journal of the American Statistical Association},&lt;br /&gt;
  volume={91},&lt;br /&gt;
  number={433},&lt;br /&gt;
  pages={217-221},&lt;br /&gt;
  year={1996},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis Group}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wang2007,&lt;br /&gt;
author = {Wang, X. and Schumitzky, A. and D'Argenio, D. Z.},&lt;br /&gt;
title = {Non linear random effects mixture models : Maximum likelihood estimation via the EM algorithm },&lt;br /&gt;
journal = {Comput. Stat. Data Anal.},&lt;br /&gt;
volume = {51},&lt;br /&gt;
pages = {6614-6623},&lt;br /&gt;
year = {2007}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Extensions &lt;br /&gt;
|linkNext=Hidden Markov models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Extensions&amp;diff=7436</id>
		<title>Extensions</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Extensions&amp;diff=7436"/>
		<updated>2013-06-25T13:28:50Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Extensions chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Extensions]]&lt;br /&gt;
*[[Extensions| Introduction ]] | [[ Mixture models ]] | [[Hidden Markov models]]  | [[Stochastic differential equations based models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;color: #2E5894; padding-left: 1.4em; padding-right:2.2em; padding-bottom:0.8em; padding:top:1&amp;quot;&amp;gt;[[Image:attention4.jpg|45px|left|link=]] &lt;br /&gt;
(If you are experiencing problems with the display of the mathematical formula, you can either try to use another browser, or use this link which should work smoothly:   http://popix.lixoft.net)&lt;br /&gt;
&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
We have so far reviewed the most frequently used models for describing both the individual parameters $(\psi_i)$ and the observations $(y_i)$, but several extensions can be considered.&lt;br /&gt;
&lt;br /&gt;
For instance,  if we assume that a population consists of several homogeneous sub-populations, mixtures models can be very useful for describing different types of mixtures, such as  mixtures of distributions, mixtures of structural models and mixtures of residual models (see [[Mixture models|Mixture models]]).&lt;br /&gt;
&lt;br /&gt;
A stochastic component can also be introduced into the model by assuming some underlying stochastic dynamics, characterized either by a [http://en.wikipedia.org/wiki/Hidden_Markov_model hidden Markov model] (see [[Hidden Markov models|Hidden Markov models]]) or  a system of [http://en.wikipedia.org/wiki/Stochastic_differential_equation stochastic differential equations] (see [[Stochastic differential equations based models]]).&lt;br /&gt;
&lt;br /&gt;
Although we restrict ourselves to these extensions in this document, it should be noted that other extensions mentioned in the introduction (see [[What is a model? A joint probability distribution!|What is a model? A joint probability distribution!]]) could also have  been addressed:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Population parameter models: introduce a priori information in an estimation context, or to model inter-population variability.&lt;br /&gt;
* Covariate models: mainly relevant in the context of wanting to simulate virtual individuals.&lt;br /&gt;
* Design models: measurement times, dose regimens, etc.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Joint models&lt;br /&gt;
|linkNext=Mixture models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Joint_models&amp;diff=7435</id>
		<title>Joint models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Joint_models&amp;diff=7435"/>
		<updated>2013-06-25T13:19:54Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Independent observations */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
An important goal of longitudinal studies is to characterize  relationships between different types of response data.&lt;br /&gt;
&lt;br /&gt;
For instance, in a PKPD population study, we may be interested in the relationship between certain pharmacokinetics (absorption, distribution, [http://en.wikipedia.org/wiki/Metabolism metabolism] and [http://en.wikipedia.org/wiki/Elimination_%28pharmacology%29 excretion]) and pharmacodynamics (biochemical and physiological effects) of a drug. To do this, we  need to measure some of both types of response data for several individuals from the same population, then try and characterize their relationship.&lt;br /&gt;
&lt;br /&gt;
Alternatively, many [http://en.wikipedia.org/wiki/Clinical_trial clinical trials] and reliability studies  generate both longitudinal and survival ([[Models for time-to-event data |time-to-event]]) data. For example, in HIV clinical trials the viral load and the concentration of [http://en.wikipedia.org/wiki/CD4%2B_cells CD4] cells are widely used as [http://en.wikipedia.org/wiki/Biomarker biomarkers] for progression to AIDS when studying the efficacy of drugs to treat HIV-infected patients. We  might then be interested in the relationship between these variables and events such as [http://en.wikipedia.org/wiki/Seroconversion seroconversion] or death.&lt;br /&gt;
&lt;br /&gt;
Therefore, in general a ''joint model'' is one that allows us to simultaneously describe the distribution of different types of observations made on the same individual. We consider this as usual in the population context.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have $L$ different types of observations for individual $i$: $y_i^{(1)}=(y_{ij}^{(1)},1\leq j \leq n_{i1})$, $y_i^{(2)}=(y_{ij}^{(2)},1\leq j \leq n_{i2})$, ..., $y_i^{(L)}=(y_{ij}^{(L)},1\leq j \leq n_{i,L})$, where $n_{i,\ell}$ is the number of observations of type $\ell$ made on individual $i$.&lt;br /&gt;
Note that $n_{i,\ell}$ may be different for different $\ell$ for the same individual, and the observation times $(t_{ij}^{(\ell)})$ too.&lt;br /&gt;
&lt;br /&gt;
Denote $y_i$ the set of observations for individual $i$:  $ y_i = (y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)})$.&lt;br /&gt;
For each individual, the joint probability distribution of the observations $y_i$  and the individual parameters $\psi_i$ can be decomposed as follows&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray} &lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pcyipsii(y_i {{!}} \psi_i) \, \ppsii(\psi_i;\theta) \\&lt;br /&gt;
&amp;amp; =&amp;amp; \pcyipsii(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)} {{!}} \psi_i) \, \ppsii(\psi_i;\theta) . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then distinguish between different types of dependency between observations: independence, conditional independence and conditional dependence.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Independent observations ==&lt;br /&gt;
&lt;br /&gt;
Suppose first that the vector of individual parameters $\psi_i$ can be decomposed into $L$ independent sub-vectors $\psi_i^{(1)}$, $\psi_i^{(2)}$, ..., $\psi_i^{(L)}$ such that $y_i^{(\ell)}$ depends only on $\psi_i^{(\ell)}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pyipsii\left(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)},\psi_i^{(1)}, \psi_i^{(2)}, \ldots , \psi_i^{(L)};\theta\right) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \prod_{\ell=1}^{L} \pmacro\left(y_i^{(\ell)},\psi_i^{(\ell)};\theta\right) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \prod_{\ell=1}^{L}  \pmacro\left(y_i^{(\ell)} {{!}} \psi_i^{(\ell)}\right) \pmacro\left(\psi_i^{(\ell)};\theta\right) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, joint modeling does not bring anything new to the picture because all information on $\psi_i^{(\ell)}$ is contained in the related set of observations $y_i^{(\ell)}$. We can therefore model separately&lt;br /&gt;
each set of observations.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A PK and PD model for [http://en.wikipedia.org/wiki/Warfarin warfarin] data&lt;br /&gt;
|text=&lt;br /&gt;
Here, 32 healthy volunteers received a 1.5 mg/kg single oral dose of warfarin, an anticoagulant normally used in the prevention of [http://en.wikipedia.org/wiki/Thrombosis thrombosis]. We then measured at different times the warfarin plasma concentration $C$  and  the [http://en.wikipedia.org/wiki/Prothrombin prothrombin] complex activity (PCA) $E$ for these patients.&lt;br /&gt;
The figure represents the PK data (on the left) and the PD data (on the right).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=warf0.png|caption= warfarin PK and PD data }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
First, we consider two entirely independent parametric models for each of the PK and PD data: a simple one compartment model $f_1$ for the PK and  rebound model $f_2$ for the PD. For any $t&amp;gt;0$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
C(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{D\, k_a}{V(k_a-k_e)} } \left( e^{-k_e \, t} - e^{-k_a \, t} \right) \\&lt;br /&gt;
E(t) &amp;amp;=&amp;amp; 100\left(\displaystyle{ \frac{\beta}{1+\beta} } e^{-\alpha \, t} + \displaystyle{ \frac{1}{1+\beta \, e^{-\gamma \, t} } }\right) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then model the observations supposing for example a combined error model for the PK data and an additive one for the PD data:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:warf1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
y_{ij}^{(1)} &amp;amp;=&amp;amp; C(t_{ij}^{(1)} ; \psi_i^{(1)}) + (a_1 + b_1\,C(t_{ij}^{(1)};\psi_i^{(1)}))\teps_{ij}^{(1)} \end{array}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:warf2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
y_{ij}^{(2)} &amp;amp;=&amp;amp; E(t_{ij}^{(2)} ; \psi_i^{(2)}) + a_2 \, \teps_{ij}^{(2)} ,  &lt;br /&gt;
\end{array}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
where $\psi_i^{(1)}=(ka_i,V_i, ke_i)$ and $\psi_i^{(2)}=(\alpha_i,\beta_i,\gamma_i)$ are independent individual parameter vectors that we suppose log-normally distributed.&lt;br /&gt;
&lt;br /&gt;
Now that the two models have been defined, we can jointly model the two data types. As they are independent, this means that we can simply use the PK model to fit the concentration data and the PD model to fit the PCA data. The figure shows the observed data and the individual predictions given by the two models for the &amp;lt;balloon title=&amp;quot;Monolix was used to fit the models. Note that the PD model is for illustrative purposes only; even though it fits well the data, it has no biological interpretation&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;4 individuals&amp;lt;/balloon&amp;gt;. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;padding-left:4em&amp;quot;&amp;gt;[[File:warfpkfit1.png|link=]]&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=warfpdfit1.png|caption=Jointly fitted PK and PD warfarin data for 4 individuals using two independent models }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In the same way that we jointly modeled these two types of independent continuous data, we can construct joint models using different types of data at the same time, i.e., various combinations of continuous, categorical, count and survival data, etc., if they are independent.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=Longitudinal and time-to-event data model&lt;br /&gt;
|text=&lt;br /&gt;
Consider the following joint model for survival and longitudinal data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; f(t_{ij} ; \psi_i^{(1)}) + g(t_{ij} ;\psi_i^{(1)})\teps_{ij} \\&lt;br /&gt;
\prob{T_i&amp;gt;t} &amp;amp;=&amp;amp; S(t ; \psi_i^{(2)}) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The continuous outcome  $y_{ij}$ and the time to event $T_i$ are independent if $\psi_i^{(1)}$ and $\psi_i^{(2)}$ are independent.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks  &lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= If the event is ''drop-out'', it is sometimes called [http://en.wikipedia.org/wiki/Missing_completely_at_random MCAR] (missing completely at random). This means that the continuous outcome does not provide any information about drop-out. }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Conditionally independent examples ==&lt;br /&gt;
&lt;br /&gt;
In this case, the various observation types depend  no longer only on disjoint (i.e., independent) individual parameters. We therefore write $\psi_i$ for the overall set of (partially or fully shared)&lt;br /&gt;
individual parameters. Observations are nevertheless supposed independent when conditioning on $\psi_i$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pyipsii(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)},\psi_i;\theta) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \left( \prod_{\ell=1}^{L} \pmacro(y_i^{(\ell)} {{!}} \psi_i)  \right) \pmacro(\psi_i;\theta) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In such cases, each observation provides information on the individual parameter vector $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
This is the most common case when we are simultaneously modeling different types of longitudinal data of the form:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij}^{(1)} &amp;amp;=&amp;amp; f_1(t_{ij}^{(1)} ; \psi_i) + g_1(t_{ij}^{(1)};\psi_i)\teps_{ij}^{(1)} \\&lt;br /&gt;
y_{ij}^{(2)} &amp;amp;=&amp;amp; f_2(t_{ij}^{(2)} ; \psi_i) + g_2(t_{ij}^{(2)};\psi_i)\teps_{ij}^{(2)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, the predictions $f_1$ and $f_2$ both depend on the same vector of individual parameters, which induces dependency between the observations $y_{i}^{(1)}$ and $y_{i}^{(2)}$. However, these observations are ''conditionally independent'' if the residual errors $\teps_{ij}^{(1)}$ and $\teps_{ij}^{(2)}$ are independent.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A joint PKPD model for  warfarin data&lt;br /&gt;
&lt;br /&gt;
|text=&lt;br /&gt;
Pertinent PKPD models aim to establish a link between a drug's concentration and its effect.&lt;br /&gt;
An indirect response model assumes that a drug does not instantaneously affect the PD response. Instead, the drug affects a precursor which then influences the PD measure. Here, as warfarin levels increase, prothrombin synthesis is inhibited, which in turn has anti-coagulant effects. Such phenomena can be approximated with a very simple ODE-based mathematical model for the PD component (we use the same one compartment model for the PK component):&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
C(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{D\, k_a}{V(k_a-k_e)} } \left( e^{-k_e \, t} - e^{-k_a \, t} \right) \\&lt;br /&gt;
E(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{k_{in} }{ k_{out} } }, \ \ \ \ t\leq 0 \\&lt;br /&gt;
\displaystyle{ \frac{d}{dt} }E(t) &amp;amp;=&amp;amp; k_{in}\left( 1 - \displaystyle{ \frac{C(t)}{IC_{50} + C(t)} } \right) - k_{out}\,E(t), \ \ \ \ t &amp;gt;0 .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We could then use the same residual error models [[#eq:warf1|(1)]] and [[#eq:warf2|(2)]] given in the previous example.&lt;br /&gt;
&lt;br /&gt;
We can also suppose that the vectors $\psi_i^{(1)}=(ka_i,V_i, ke_i)$ and $\psi_i^{(2)}=(IC_{50,i},k_{in,i},k_{out,i})$ are independent, but the fact that the effect $E$ predicted by the model is a function&lt;br /&gt;
of the concentration $C$ introduces dependence between the two observation types because both depend on the PK parameters $\psi_i^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
If the residual errors $(\teps_{ij}^{(1)})$ and $(\teps_{ij}^{(2)})$ are independent, then the observations are conditionally independent, i.e., when the predicted concentration $C(t)$ is given, the observed concentrations $\by^{(1)}$ do not bring any further information on the distribution of the PD observations $\by^{(2)}$.&lt;br /&gt;
&lt;br /&gt;
This joint model can be used to model the same warfarin data as before (again, using $\monolix$).&lt;br /&gt;
The figure shows the resulting individual predictions.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;margin-left:4.2em&amp;quot;&amp;gt;[[File:warfpkfit2.png|link=]]&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=warfpdfit2.png|caption=Fitted PK and PD warfarin data for 4 individuals using a conditionally independent joint model}}&lt;br /&gt;
}}&lt;br /&gt;
 &lt;br /&gt;
&lt;br /&gt;
We can extend this framework to different types of data, considering for example categorical observations  $y_i^{(2)}$ for which the probabilities $\prob{y_{ij}^{(2)} = k}$ depend on  $f_1(t_{ij}^{(2)};\psi_i)$ and consequently $\psi_i$. We can also consider survival data for which the risk function depends on $f_1$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=Longitudinal and time-to-event data model&lt;br /&gt;
&lt;br /&gt;
|text=Consider a joint model for survival and longitudinal data, assuming now that the hazard function (or equivalently the survival function) depends on the continuous data prediction:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; f(t_{ij} ; \psi_i) + g(t_{ij} ;\psi_i)\teps_{ij} \\&lt;br /&gt;
\prob{T_i&amp;gt;t} &amp;amp;=&amp;amp; S(t ; f(t ; \psi_i)) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
If for instance $(y_{ij})$ is the measured viral load of an HIV infected patient, we can assume that the probability of events such as death,  seroconversion or drop-out depends on the &amp;quot;true&amp;quot; viral load  $f(t ; \psi_i)$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=  if the event is ''drop-out'', it is sometimes called MAR (missing at random). This means that the probability of drop-out depends on some of the individual parameters, but that the observation itself of the continuous outcome does not provide any additional information. In our example, this means that the probability that a patient leaves the study depends on their true state (i.e., their true but unknown viral load), and not on the measured  viral load values. }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Conditionally dependent observations ==&lt;br /&gt;
&lt;br /&gt;
In this case, there is a dependency structure between types of observation  that no longer allows us to decompose the joint model into a product of models with only one type of observation in each.&lt;br /&gt;
&lt;br /&gt;
This kind of dependency occurs when several types of longitudinal data are obtained at the same times, with correlated measurement errors. The joint conditional distribution $\qcyipsii$ of the observations is&lt;br /&gt;
Gaussian if the residual errors are. The dependency structure between observations can then be characterized by a variance-covariance matrix for the errors.&lt;br /&gt;
&lt;br /&gt;
We can also consider a natural decomposition of this joint distribution into a product of conditional distributions:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pyipsii(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)},\psi_i;\theta) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \pmacro(y_i^{(1)} {{!}} \psi_i;\theta) \pmacro(y_i^{(2)} {{!}} y_i^{(1)}, \psi_i;\theta)\ldots \pmacro(y_i^{(L)} {{!}} y_i^{(1)},\ldots,y_i^{(L-1)}, \psi_i;\theta) \pmacro(\psi_i;\theta) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, the distribution of $y_i^{(2)}$ depends on the observation $y_i^{(1)}$, the distribution of $y_i^{(3)}$ depends on $y_i^{(1)}$ and $y_i^{(2)}$, etc.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A longitudinal data and drop-out model&lt;br /&gt;
&lt;br /&gt;
|text= Consider a joint model for longitudinal data and drop-out, assuming now that the hazard function (or equivalently the survival function) depends on the observed data itself:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; f(t_{ij} ; \psi_i) + g(t_{ij} ;\psi_i)\teps_{ij} \\&lt;br /&gt;
\prob{T_i&amp;gt;t} &amp;amp;=&amp;amp; S(t ; (y_{ij}, t_{ij}&amp;lt;t)) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This drop-out  mechanism is sometimes called MNAR (missing not at random).&lt;br /&gt;
In this example where $(y_{ij}, t_{ij}&amp;lt;t)$ is the sequence of measured viral loads before time $t$, MNAR means that the probability that a patient leaves the study depends on their previously-measured viral concentrations. &lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--&lt;br /&gt;
== $\mlxtran$ for joint models==&lt;br /&gt;
&lt;br /&gt;
TO DO&lt;br /&gt;
--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{albert2004modeling,&lt;br /&gt;
  title={Modeling repeated count data subject to informative dropout},&lt;br /&gt;
  author={Albert, P. S. and Follmann, D. A.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={56},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={667-677},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{chi2006joint,&lt;br /&gt;
  title={Joint models for multivariate longitudinal and multivariate survival data},&lt;br /&gt;
  author={Chi, Y.-Y. and Ibrahim, J. G.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={62},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={432-445},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{de1994modelling,&lt;br /&gt;
  title={Modelling progression of CD4-lymphocyte count and its relationship to survival time},&lt;br /&gt;
  author={De Gruttola, V. and Tu, X. M.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  pages={1003-1014},&lt;br /&gt;
  year={1994},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{henderson2000joint,&lt;br /&gt;
  title={Joint modelling of longitudinal measurements and event time data.},&lt;br /&gt;
  author={Henderson, R. and Diggle, P. and Dobson, A.},&lt;br /&gt;
  journal={Biostatistics},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={465-480},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Biometrika Trust}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hsieh2006joint,&lt;br /&gt;
  title={Joint modeling of survival and longitudinal data: likelihood approach revisited},&lt;br /&gt;
  author={Hsieh, F. and Tseng, Y.-K. and Wang, J.-L.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={62},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={1037-1043},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hu2003joint,&lt;br /&gt;
  title={A joint model for nonlinear longitudinal data with informative dropout},&lt;br /&gt;
  author={Hu, C. and Sale, M. E.},&lt;br /&gt;
  journal={Journal of pharmacokinetics and pharmacodynamics},&lt;br /&gt;
  volume={30},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={83-103},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{liu2009,&lt;br /&gt;
author = {Liu, L.  and Huang, X. },&lt;br /&gt;
title = {Joint analysis of correlated repeated measures and recurrent events processes in the presence of a dependent terminal event},&lt;br /&gt;
journal = {J. ROY. STAT. SOC. C-APP.},&lt;br /&gt;
volume = {58},&lt;br /&gt;
pages = {65-81},&lt;br /&gt;
year = {2009}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{rizopoulos2012,&lt;br /&gt;
        author = {Rizopoulos, D. },&lt;br /&gt;
        title  = {Joint Models for Longitudinal and Time-to-Event Data. With Applications in R.},&lt;br /&gt;
        publisher = {Chapman &amp;amp; Hall/CRC Biostatistics},&lt;br /&gt;
        address = {Boca Raton},&lt;br /&gt;
        year = {2012}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{rondeau2007,&lt;br /&gt;
author = {Rondeau, V.  and  Mathoulin-Pelissier, S. and Jacqmin-Gadda, H.  and Brouste, V.  and Soubeyran, P. },&lt;br /&gt;
title  = {Joint frailty models for recurring events and death using maximum penalized likelihood estimation: application on cancer events.},&lt;br /&gt;
journal = {Biostatistics},&lt;br /&gt;
volume = {8},&lt;br /&gt;
pages = {708-721},&lt;br /&gt;
year = {2007}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibitex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{song2004semiparametric,&lt;br /&gt;
  title={A Semiparametric Likelihood Approach to Joint Modeling of Longitudinal and Time-to-Event Data},&lt;br /&gt;
  author={Song,X. and Davidian,M. and Tsiatis,A. A.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={58},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={742-753},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{tsiatis2004joint,&lt;br /&gt;
  title={Joint modeling of longitudinal and time-to-event data: an overview},&lt;br /&gt;
  author={Tsiatis, A. A. and Davidian, M.},&lt;br /&gt;
  journal={Statistica Sinica},&lt;br /&gt;
  volume={14},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={809-834},&lt;br /&gt;
  year={2004}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wu2002joint,&lt;br /&gt;
  title={A joint model for nonlinear mixed-effects models with censoring and covariates measured with error, with application to AIDS studies},&lt;br /&gt;
  author={Wu, L.},&lt;br /&gt;
  journal={Journal of the American Statistical association},&lt;br /&gt;
  volume={97},&lt;br /&gt;
  number={460},&lt;br /&gt;
  pages={955-964},&lt;br /&gt;
  year={2002},&lt;br /&gt;
  publisher={American Statistical Association}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wulfsohn1997joint,&lt;br /&gt;
  title={A joint model for survival and longitudinal data measured with error},&lt;br /&gt;
  author={Wulfsohn, M. S. and Tsiatis, A. A.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  pages={330-339},&lt;br /&gt;
  year={1997},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkNext=Extensions&lt;br /&gt;
|linkBack=Models for time-to-event data  }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Joint_models&amp;diff=7434</id>
		<title>Joint models</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Joint_models&amp;diff=7434"/>
		<updated>2013-06-25T13:15:32Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
==Introduction==&lt;br /&gt;
&lt;br /&gt;
An important goal of longitudinal studies is to characterize  relationships between different types of response data.&lt;br /&gt;
&lt;br /&gt;
For instance, in a PKPD population study, we may be interested in the relationship between certain pharmacokinetics (absorption, distribution, [http://en.wikipedia.org/wiki/Metabolism metabolism] and [http://en.wikipedia.org/wiki/Elimination_%28pharmacology%29 excretion]) and pharmacodynamics (biochemical and physiological effects) of a drug. To do this, we  need to measure some of both types of response data for several individuals from the same population, then try and characterize their relationship.&lt;br /&gt;
&lt;br /&gt;
Alternatively, many [http://en.wikipedia.org/wiki/Clinical_trial clinical trials] and reliability studies  generate both longitudinal and survival ([[Models for time-to-event data |time-to-event]]) data. For example, in HIV clinical trials the viral load and the concentration of [http://en.wikipedia.org/wiki/CD4%2B_cells CD4] cells are widely used as [http://en.wikipedia.org/wiki/Biomarker biomarkers] for progression to AIDS when studying the efficacy of drugs to treat HIV-infected patients. We  might then be interested in the relationship between these variables and events such as [http://en.wikipedia.org/wiki/Seroconversion seroconversion] or death.&lt;br /&gt;
&lt;br /&gt;
Therefore, in general a ''joint model'' is one that allows us to simultaneously describe the distribution of different types of observations made on the same individual. We consider this as usual in the population context.&lt;br /&gt;
&lt;br /&gt;
Suppose that we have $L$ different types of observations for individual $i$: $y_i^{(1)}=(y_{ij}^{(1)},1\leq j \leq n_{i1})$, $y_i^{(2)}=(y_{ij}^{(2)},1\leq j \leq n_{i2})$, ..., $y_i^{(L)}=(y_{ij}^{(L)},1\leq j \leq n_{i,L})$, where $n_{i,\ell}$ is the number of observations of type $\ell$ made on individual $i$.&lt;br /&gt;
Note that $n_{i,\ell}$ may be different for different $\ell$ for the same individual, and the observation times $(t_{ij}^{(\ell)})$ too.&lt;br /&gt;
&lt;br /&gt;
Denote $y_i$ the set of observations for individual $i$:  $ y_i = (y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)})$.&lt;br /&gt;
For each individual, the joint probability distribution of the observations $y_i$  and the individual parameters $\psi_i$ can be decomposed as follows&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray} &lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pcyipsii(y_i {{!}} \psi_i) \, \ppsii(\psi_i;\theta) \\&lt;br /&gt;
&amp;amp; =&amp;amp; \pcyipsii(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)} {{!}} \psi_i) \, \ppsii(\psi_i;\theta) . &lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then distinguish between different types of dependency between observations: independence, conditional independence and conditional dependence.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Independent observations ==&lt;br /&gt;
&lt;br /&gt;
Suppose first that the vector of individual parameters $\psi_i$ can be decomposed into $L$ independent sub-vectors $\psi_i^{(1)}$, $\psi_i^{(2)}$, ..., $\psi_i^{(L)}$ such that $y_i^{(\ell)}$ depends only on $\psi_i^{(\ell)}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pyipsii\left(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)},\psi_i^{(1)}, \psi_i^{(2)}, \ldots , \psi_i^{(L)};\theta\right) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \prod_{\ell=1}^{L} \pmacro\left(y_i^{(\ell)},\psi_i^{(\ell)};\theta\right) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \prod_{\ell=1}^{L}  \pmacro\left(y_i^{(\ell)} {{!}} \psi_i^{(\ell)}\right) \pmacro\left(\psi_i^{(\ell)};\theta\right) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, joint modeling does not bring anything new to the picture because all information on $\psi_i^{(\ell)}$ is contained in the related set of observations $y_i^{(\ell)}$. We can therefore model separately&lt;br /&gt;
each set of observations.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A PK and PD model for warfarin data&lt;br /&gt;
|text=&lt;br /&gt;
Here, 32 healthy volunteers received a 1.5 mg/kg single oral dose of warfarin, an anticoagulant normally used in the prevention of thrombosis.&lt;br /&gt;
We then measured at different times the warfarin plasma concentration $C$  and  the prothrombin complex activity (PCA) $E$ for these patients.&lt;br /&gt;
The figure represents the PK data (on the left) and the PD data (on the right).&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=warf0.png|caption= warfarin PK and PD data }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
First, we consider two entirely independent parametric models for each of the PK and PD data: a simple one compartment model $f_1$ for the PK and  rebound model $f_2$ for the PD. For any $t&amp;gt;0$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
C(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{D\, k_a}{V(k_a-k_e)} } \left( e^{-k_e \, t} - e^{-k_a \, t} \right) \\&lt;br /&gt;
E(t) &amp;amp;=&amp;amp; 100\left(\displaystyle{ \frac{\beta}{1+\beta} } e^{-\alpha \, t} + \displaystyle{ \frac{1}{1+\beta \, e^{-\gamma \, t} } }\right) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We can then model the observations supposing for example a combined error model for the PK data and an additive one for the PD data:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:warf1&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
y_{ij}^{(1)} &amp;amp;=&amp;amp; C(t_{ij}^{(1)} ; \psi_i^{(1)}) + (a_1 + b_1\,C(t_{ij}^{(1)};\psi_i^{(1)}))\teps_{ij}^{(1)} \end{array}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt; &lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;eq:warf2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;\begin{array}{c}&lt;br /&gt;
y_{ij}^{(2)} &amp;amp;=&amp;amp; E(t_{ij}^{(2)} ; \psi_i^{(2)}) + a_2 \, \teps_{ij}^{(2)} ,  &lt;br /&gt;
\end{array}&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
where $\psi_i^{(1)}=(ka_i,V_i, ke_i)$ and $\psi_i^{(2)}=(\alpha_i,\beta_i,\gamma_i)$ are independent individual parameter vectors that we suppose log-normally distributed.&lt;br /&gt;
&lt;br /&gt;
Now that the two models have been defined, we can jointly model the two data types. As they are independent, this means that we can simply use the PK model to fit the concentration data&lt;br /&gt;
and the PD model to fit the PCA data. The figure shows the observed data and the individual predictions given by the two models for the &amp;lt;balloon title=&amp;quot;Monolix was used to fit the models. Note that the PD model is for illustrative purposes only; even though it fits well the data, it has no biological interpretation&amp;quot; style=&amp;quot;color:#177245&amp;quot;&amp;gt;4 individuals&amp;lt;/balloon&amp;gt;. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;padding-left:4em&amp;quot;&amp;gt;[[File:warfpkfit1.png|link=]]&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=warfpdfit1.png|caption=Jointly fitted PK and PD warfarin data for 4 individuals using two independent models }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
In the same way that we jointly modeled these two types of independent continuous data, we can construct joint models using different types of data at the same time, i.e., various combinations of continuous, categorical, count and survival data, etc., if they are independent.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=Longitudinal and time-to-event data model&lt;br /&gt;
|text=&lt;br /&gt;
Consider the following joint model for survival and longitudinal data:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; f(t_{ij} ; \psi_i^{(1)}) + g(t_{ij} ;\psi_i^{(1)})\teps_{ij} \\&lt;br /&gt;
\prob{T_i&amp;gt;t} &amp;amp;=&amp;amp; S(t ; \psi_i^{(2)}) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The continuous outcome  $y_{ij}$ and the time to event $T_i$ are independent if $\psi_i^{(1)}$ and $\psi_i^{(2)}$ are independent.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks  &lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= If the event is ''drop-out'', it is sometimes called MCAR (missing completely at random). This means that the continuous outcome does not provide any information about drop-out. }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Conditionally independent examples ==&lt;br /&gt;
&lt;br /&gt;
In this case, the various observation types depend  no longer only on disjoint (i.e., independent) individual parameters. We therefore write $\psi_i$ for the overall set of (partially or fully shared)&lt;br /&gt;
individual parameters. Observations are nevertheless supposed independent when conditioning on $\psi_i$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pyipsii(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)},\psi_i;\theta) \\&lt;br /&gt;
&amp;amp;=&amp;amp; \left( \prod_{\ell=1}^{L} \pmacro(y_i^{(\ell)} {{!}} \psi_i)  \right) \pmacro(\psi_i;\theta) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
In such cases, each observation provides information on the individual parameter vector $\psi_i$.&lt;br /&gt;
&lt;br /&gt;
This is the most common case when we are simultaneously modeling different types of longitudinal data of the form:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij}^{(1)} &amp;amp;=&amp;amp; f_1(t_{ij}^{(1)} ; \psi_i) + g_1(t_{ij}^{(1)};\psi_i)\teps_{ij}^{(1)} \\&lt;br /&gt;
y_{ij}^{(2)} &amp;amp;=&amp;amp; f_2(t_{ij}^{(2)} ; \psi_i) + g_2(t_{ij}^{(2)};\psi_i)\teps_{ij}^{(2)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, the predictions $f_1$ and $f_2$ both depend on the same vector of individual parameters, which induces dependency between the observations $y_{i}^{(1)}$ and $y_{i}^{(2)}$. However, these observations are ''conditionally independent'' if the residual errors $\teps_{ij}^{(1)}$ and $\teps_{ij}^{(2)}$ are independent.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A joint PKPD model for  warfarin data&lt;br /&gt;
&lt;br /&gt;
|text=&lt;br /&gt;
Pertinent PKPD models aim to establish a link between a drug's concentration and its effect.&lt;br /&gt;
An indirect response model assumes that a drug does not instantaneously affect the PD response. Instead, the drug affects a precursor which then influences the PD measure. Here, as warfarin levels increase, prothrombin synthesis is inhibited, which in turn has anti-coagulant effects. Such phenomena can be approximated with a very simple ODE-based mathematical model for the PD component (we use the same one compartment model for the PK component):&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
C(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{D\, k_a}{V(k_a-k_e)} } \left( e^{-k_e \, t} - e^{-k_a \, t} \right) \\&lt;br /&gt;
E(t) &amp;amp;=&amp;amp; \displaystyle{ \frac{k_{in} }{ k_{out} } }, \ \ \ \ t\leq 0 \\&lt;br /&gt;
\displaystyle{ \frac{d}{dt} }E(t) &amp;amp;=&amp;amp; k_{in}\left( 1 - \displaystyle{ \frac{C(t)}{IC_{50} + C(t)} } \right) - k_{out}\,E(t), \ \ \ \ t &amp;gt;0 .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
We could then use the same residual error models [[#eq:warf1|(1)]] and [[#eq:warf2|(2)]] given in the previous example.&lt;br /&gt;
&lt;br /&gt;
We can also suppose that the vectors $\psi_i^{(1)}=(ka_i,V_i, ke_i)$ and $\psi_i^{(2)}=(IC_{50,i},k_{in,i},k_{out,i})$ are independent, but the fact that the effect $E$ predicted by the model is a function&lt;br /&gt;
of the concentration $C$ introduces dependence between the two observation types because both depend on the PK parameters $\psi_i^{(1)}$.&lt;br /&gt;
&lt;br /&gt;
If the residual errors $(\teps_{ij}^{(1)})$ and $(\teps_{ij}^{(2)})$ are independent, then the observations are conditionally independent, i.e., when the predicted concentration $C(t)$ is given, the observed concentrations $\by^{(1)}$ do not bring any further information on the distribution of the PD observations $\by^{(2)}$.&lt;br /&gt;
&lt;br /&gt;
This joint model can be used to model the same warfarin data as before (again, using $\monolix$).&lt;br /&gt;
The figure shows the resulting individual predictions.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;margin-left:4.2em&amp;quot;&amp;gt;[[File:warfpkfit2.png|link=]]&amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ImageWithCaption|image=warfpdfit2.png|caption=Fitted PK and PD warfarin data for 4 individuals using a conditionally independent joint model}}&lt;br /&gt;
}}&lt;br /&gt;
 &lt;br /&gt;
&lt;br /&gt;
We can extend this framework to different types of data, considering for example categorical observations  $y_i^{(2)}$ for which the probabilities $\prob{y_{ij}^{(2)} = k}$ depend on  $f_1(t_{ij}^{(2)};\psi_i)$ and consequently $\psi_i$. We can also consider survival data for which the risk function depends on $f_1$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=Longitudinal and time-to-event data model&lt;br /&gt;
&lt;br /&gt;
|text=Consider a joint model for survival and longitudinal data, assuming now that the hazard function (or equivalently the survival function) depends on the continuous data prediction:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; f(t_{ij} ; \psi_i) + g(t_{ij} ;\psi_i)\teps_{ij} \\&lt;br /&gt;
\prob{T_i&amp;gt;t} &amp;amp;=&amp;amp; S(t ; f(t ; \psi_i)) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
If for instance $(y_{ij})$ is the measured viral load of an HIV infected patient, we can assume that the probability of events such as death,  seroconversion or drop-out depends on the &amp;quot;true&amp;quot; viral load  $f(t ; \psi_i)$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text=  if the event is ''drop-out'', it is sometimes called MAR (missing at random). This means that the probability of drop-out depends on some of the individual parameters, but that the observation itself of the continuous outcome does not provide any additional information. In our example, this means that the probability that a patient leaves the study depends on their true state (i.e., their true but unknown viral load), and not on the measured  viral load values. }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Conditionally dependent observations ==&lt;br /&gt;
&lt;br /&gt;
In this case, there is a dependency structure between types of observation  that no longer allows us to decompose the joint model into a product of models with only one type of observation in each.&lt;br /&gt;
&lt;br /&gt;
This kind of dependency occurs when several types of longitudinal data are obtained at the same times, with correlated measurement errors. The joint conditional distribution $\qcyipsii$ of the observations is&lt;br /&gt;
Gaussian if the residual errors are. The dependency structure between observations can then be characterized by a variance-covariance matrix for the errors.&lt;br /&gt;
&lt;br /&gt;
We can also consider a natural decomposition of this joint distribution into a product of conditional distributions:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pyipsii(y_i,\psi_i;\theta) &amp;amp;=&amp;amp; \pyipsii(y_i^{(1)},y_i^{(2)},\ldots,y_i^{(L)},\psi_i;\theta) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \pmacro(y_i^{(1)} {{!}} \psi_i;\theta) \pmacro(y_i^{(2)} {{!}} y_i^{(1)}, \psi_i;\theta)\ldots \pmacro(y_i^{(L)} {{!}} y_i^{(1)},\ldots,y_i^{(L-1)}, \psi_i;\theta) \pmacro(\psi_i;\theta) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Here, the distribution of $y_i^{(2)}$ depends on the observation $y_i^{(1)}$, the distribution of $y_i^{(3)}$ depends on $y_i^{(1)}$ and $y_i^{(2)}$, etc.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example1&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=A longitudinal data and drop-out model&lt;br /&gt;
&lt;br /&gt;
|text= Consider a joint model for longitudinal data and drop-out, assuming now that the hazard function (or equivalently the survival function) depends on the observed data itself:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
y_{ij} &amp;amp;=&amp;amp; f(t_{ij} ; \psi_i) + g(t_{ij} ;\psi_i)\teps_{ij} \\&lt;br /&gt;
\prob{T_i&amp;gt;t} &amp;amp;=&amp;amp; S(t ; (y_{ij}, t_{ij}&amp;lt;t)) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
This drop-out  mechanism is sometimes called MNAR (missing not at random).&lt;br /&gt;
In this example where $(y_{ij}, t_{ij}&amp;lt;t)$ is the sequence of measured viral loads before time $t$, MNAR means that the probability that a patient leaves the study depends on their previously-measured viral concentrations. &lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--&lt;br /&gt;
== $\mlxtran$ for joint models==&lt;br /&gt;
&lt;br /&gt;
TO DO&lt;br /&gt;
--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{albert2004modeling,&lt;br /&gt;
  title={Modeling repeated count data subject to informative dropout},&lt;br /&gt;
  author={Albert, P. S. and Follmann, D. A.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={56},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={667-677},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{chi2006joint,&lt;br /&gt;
  title={Joint models for multivariate longitudinal and multivariate survival data},&lt;br /&gt;
  author={Chi, Y.-Y. and Ibrahim, J. G.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={62},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={432-445},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{de1994modelling,&lt;br /&gt;
  title={Modelling progression of CD4-lymphocyte count and its relationship to survival time},&lt;br /&gt;
  author={De Gruttola, V. and Tu, X. M.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  pages={1003-1014},&lt;br /&gt;
  year={1994},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{henderson2000joint,&lt;br /&gt;
  title={Joint modelling of longitudinal measurements and event time data.},&lt;br /&gt;
  author={Henderson, R. and Diggle, P. and Dobson, A.},&lt;br /&gt;
  journal={Biostatistics},&lt;br /&gt;
  volume={1},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={465-480},&lt;br /&gt;
  year={2000},&lt;br /&gt;
  publisher={Biometrika Trust}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hsieh2006joint,&lt;br /&gt;
  title={Joint modeling of survival and longitudinal data: likelihood approach revisited},&lt;br /&gt;
  author={Hsieh, F. and Tseng, Y.-K. and Wang, J.-L.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={62},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={1037-1043},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hu2003joint,&lt;br /&gt;
  title={A joint model for nonlinear longitudinal data with informative dropout},&lt;br /&gt;
  author={Hu, C. and Sale, M. E.},&lt;br /&gt;
  journal={Journal of pharmacokinetics and pharmacodynamics},&lt;br /&gt;
  volume={30},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={83-103},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{liu2009,&lt;br /&gt;
author = {Liu, L.  and Huang, X. },&lt;br /&gt;
title = {Joint analysis of correlated repeated measures and recurrent events processes in the presence of a dependent terminal event},&lt;br /&gt;
journal = {J. ROY. STAT. SOC. C-APP.},&lt;br /&gt;
volume = {58},&lt;br /&gt;
pages = {65-81},&lt;br /&gt;
year = {2009}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{rizopoulos2012,&lt;br /&gt;
        author = {Rizopoulos, D. },&lt;br /&gt;
        title  = {Joint Models for Longitudinal and Time-to-Event Data. With Applications in R.},&lt;br /&gt;
        publisher = {Chapman &amp;amp; Hall/CRC Biostatistics},&lt;br /&gt;
        address = {Boca Raton},&lt;br /&gt;
        year = {2012}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{rondeau2007,&lt;br /&gt;
author = {Rondeau, V.  and  Mathoulin-Pelissier, S. and Jacqmin-Gadda, H.  and Brouste, V.  and Soubeyran, P. },&lt;br /&gt;
title  = {Joint frailty models for recurring events and death using maximum penalized likelihood estimation: application on cancer events.},&lt;br /&gt;
journal = {Biostatistics},&lt;br /&gt;
volume = {8},&lt;br /&gt;
pages = {708-721},&lt;br /&gt;
year = {2007}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibitex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{song2004semiparametric,&lt;br /&gt;
  title={A Semiparametric Likelihood Approach to Joint Modeling of Longitudinal and Time-to-Event Data},&lt;br /&gt;
  author={Song,X. and Davidian,M. and Tsiatis,A. A.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={58},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={742-753},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{tsiatis2004joint,&lt;br /&gt;
  title={Joint modeling of longitudinal and time-to-event data: an overview},&lt;br /&gt;
  author={Tsiatis, A. A. and Davidian, M.},&lt;br /&gt;
  journal={Statistica Sinica},&lt;br /&gt;
  volume={14},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={809-834},&lt;br /&gt;
  year={2004}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wu2002joint,&lt;br /&gt;
  title={A joint model for nonlinear mixed-effects models with censoring and covariates measured with error, with application to AIDS studies},&lt;br /&gt;
  author={Wu, L.},&lt;br /&gt;
  journal={Journal of the American Statistical association},&lt;br /&gt;
  volume={97},&lt;br /&gt;
  number={460},&lt;br /&gt;
  pages={955-964},&lt;br /&gt;
  year={2002},&lt;br /&gt;
  publisher={American Statistical Association}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wulfsohn1997joint,&lt;br /&gt;
  title={A joint model for survival and longitudinal data measured with error},&lt;br /&gt;
  author={Wulfsohn, M. S. and Tsiatis, A. A.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  pages={330-339},&lt;br /&gt;
  year={1997},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkNext=Extensions&lt;br /&gt;
|linkBack=Models for time-to-event data  }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_time-to-event_data&amp;diff=7433</id>
		<title>Models for time-to-event data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_time-to-event_data&amp;diff=7433"/>
		<updated>2013-06-25T13:10:21Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Repeated events */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, observations are the &amp;quot;times at which events occur&amp;quot;. An event may be one-off (e.g., death, hardware failure) or repeated (e.g., epileptic seizures, metro strike).&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Single event==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
To begin with, we will consider a one-off event.&lt;br /&gt;
Depending on the application, the length of time to this event  may be called the ''survival'' time (until death), ''failure'' time (until hardware fails), etc. To be general, we can just say ''event'' time.&lt;br /&gt;
&lt;br /&gt;
The random variable representing the event time for subject $i$ is typically written $T_i$. Several situations are then possible to define the observations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The event time is exactly observed.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: Then, the observation for individual $i$ is $y_i = t_i$, where $t_i$ is a realization of the random variable $T_i$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* We may know the event has happened in an interval $I_i$ but not know the exact time $t_i$. This is ''interval censoring''. For example, at a routine check-up, cancer recurrence may be detected, and we only know that it has occurred at some point in time since the last check-up.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival3.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: The observation for individual $i$ is the event: $y_i = $ &amp;quot;$a_i &amp;lt; t_i \leq b_i$&amp;quot;.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If we assume that the trial ends at time $\tstop$, then the event may happen after the end of the trial period. This is ''right censoring''.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: There are several variations of this for defining what the observations are:&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If events (before $\tstop$) are exactly observed, then for $i=1,2,\ldots, N$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1|&lt;br /&gt;
equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_i = \left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
t_i &amp;amp; {\rm if \quad} t_i \leq \tstop \\&lt;br /&gt;
{\rm t_i &amp;gt; \tstop \quad} &amp;amp; {\rm otherwise. \quad}&lt;br /&gt;
\end{array} \right.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithText&amp;amp;Table&lt;br /&gt;
|title1=Example:&lt;br /&gt;
|title2=&lt;br /&gt;
|equation=&lt;br /&gt;
Assume  that a trial starts at $\tstart=0$ and ends at $\tstop=5$, and that we obtain the following observations from 4 individuals: &lt;br /&gt;
&lt;br /&gt;
$y_1 = 3.2$ &lt;br /&gt;
&lt;br /&gt;
$y_2=$ &amp;quot;$t_2&amp;gt;5$&amp;quot;&lt;br /&gt;
&lt;br /&gt;
$y_3= 2.7$ &lt;br /&gt;
&lt;br /&gt;
$y_4 =$ &amp;quot;$t_4&amp;gt;5$&amp;quot;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These observations can be stored in a data file as shown in the table on the right.&lt;br /&gt;
&lt;br /&gt;
Here, &amp;quot;event=0&amp;quot; at time $t$ means that the event happened after $t$ while &amp;quot;event=1&amp;quot; means that the event happened at time $t$. &lt;br /&gt;
&lt;br /&gt;
The lines with $t=0$ are used to state the trial start time $\tstart=0$.&lt;br /&gt;
&lt;br /&gt;
|table=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 75%&amp;quot;&lt;br /&gt;
!{{!}} ID {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3.2 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 2.7 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}} &lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* If events before $\tstop$ are interval censored, then for $i=1,2,\ldots, N$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_i = \left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
{\rm a_i &amp;lt; t_i \quad \leq \quad b_i} &amp;amp; {\rm if \quad} t_i\leq \tstop \\&lt;br /&gt;
{\rm t_i  &amp;gt;  \tstop \quad} &amp;amp; {\rm otherwise.}&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithText&amp;amp;Table&lt;br /&gt;
|title1=Example:&lt;br /&gt;
|title2=&lt;br /&gt;
|equation=&lt;br /&gt;
Assume that we have censoring intervals of length 1: &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
$(0,1],(1,2],\ldots,(4,5]$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the same four individuals as the previous example, we now have the following observations: &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
$y_1=$  &amp;quot;$3 &amp;lt; t_1 \leq 4$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_2=$  &amp;quot;$t_2&amp;gt;5$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_3=$ &amp;quot;$2&amp;lt; t_3 \leq 3$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_4=$  &amp;quot;$t_4&amp;gt;5$&amp;quot;. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These observations can be stored in a data file as shown in the table on the right.&lt;br /&gt;
&lt;br /&gt;
Here &amp;quot;event=0&amp;quot; at time $t$ means that the event happened after $t$ while &amp;quot;event=1&amp;quot; means that the event happened before time $t$.&lt;br /&gt;
|table=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 75%&amp;quot;&lt;br /&gt;
!{{!}} ID {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 2 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 3 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Probability distributions == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Several functions play key roles in time-to-event analysis: the [http://en.wikipedia.org/wiki/Survival_function survival function] the [http://en.wikipedia.org/wiki/Survival_analysis#Hazard_function_and_cumulative_hazard_function hazard function] and the [http://en.wikipedia.org/wiki/Survival_analysis#Hazard_function_and_cumulative_hazard_function cumulative hazard function].&lt;br /&gt;
We are still working under a population approach here and so these functions, detailed below, are therefore individual functions, i.e., each subject has its own. As we are using parametric models, this means that these functions depend on individual parameters $(\psi_i)$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The '''survival function''' $S(t; \psi_i)$ gives the probability that the event happens to individual $i$ after time $t&amp;gt;t_{start}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
S(t; \psi_i) \ \ \eqdef \ \ \prob{T_i&amp;gt;t ; \psi_i} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* The '''hazard function''' $\hazard(t;\psi_i)$ is defined for individual $i$ as the instantaneous rate of the event at time $t$, given that the event has not already occurred:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;  &lt;br /&gt;
\hazard(t;\psi_i) \ \ \eqdef \ \  \lim_{dt\to 0} \displaystyle{\frac{S(t;\psi_i) - S(t + dt;\psi_i)}{ S(t;\psi_i) \, dt} }. &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: This is equivalent to: &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;HazardSurvival&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\hazard(t;\psi_i) \ \ = \ \ -\displaystyle{ \frac{d}{dt} } \log{S(t;\psi_i)}. &lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1)&lt;br /&gt;
}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Another useful quantity is the '''cumulative hazard function''' $\cumhaz(a,b;\psi_i)$,  defined for individual $i$ as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\cumhaz(a,b;\psi_i) \ \ \eqdef \ \ \displaystyle{\int_a^b \hazard(t;\psi_i) \, dt }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
: Note that [[#HazardSurvival|(1)]] implies that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
S(t;\psi_i) \ \ = \ \ e^{-\cumhaz(t_{start},t;\psi_i)}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Equation [[#HazardSurvival|(1)]] shows that the hazard function $\hazard(t;\psi_i)$ characterizes the problem, because knowing it is the same as knowing the survival function $S(t;\psi_i)$. The probability distribution of survival data is therefore completely defined by the hazard function.&lt;br /&gt;
Let $\qcyipsii$ be the conditional distribution of the  observation $y_i$ given the vector of individual parameters $\psi_i$. Its pdf can be easily computed for the various censoring situations discussed above:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is exactly observed with $y_i=t_i$, the density is the derivative of the cumulative density function, i.e., the derivative of $1 - S(t_i;\psi_i)$:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\begin{eqnarray}\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \frac{d}{dt_i}\left(1 - e^{-\cumhaz(t_{start},t_i;\psi_i)}\right)\\&lt;br /&gt;
%&amp;amp;=&amp;amp; \left(\frac{d}{dt_i} \int_{t_{start} }^{t_i} \hazard(u;\psi_i) \, du   \right)  e^{-\cumhaz(t_{start},t_i;\psi_i)}\\&lt;br /&gt;
&amp;amp;=&amp;amp;\hazard(t_i;\psi_i)e^{-\cumhaz(t_{start},t_i;\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is interval-censored with $y_i=\,$  &amp;quot;$a_i&amp;lt;t_i\leq b_i$&amp;quot;:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \prob{T_i \in (a_i,b_i]\,{{!}} \,\psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; \prob{T_i \leq b_i {{!}} \psi_i} - \prob{T_i \leq a_i {{!}} \psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; (1-S( b_i ; \psi_i)) - (1-S( a_i ; \psi_i)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{start},a_i;\psi_i)} - e^{-\cumhaz(t_{start},b_i;\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is right-censored with $y_i= \,$ &amp;quot;$t_i&amp;gt;t_{stop}$&amp;quot;:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \prob{T_i &amp;gt; t_{stop}   {{!}} \psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; S( t_{stop} ; \psi_i) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{start},t_{stop};\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Repeated events==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Sometimes, an event can potentially happen again and again, e.g., [http://en.wikipedia.org/wiki/Epileptic_seizure epileptic seizures], heart attacks,  etc.&lt;br /&gt;
For any given hazard function $\hazard$, the survival function $S$ for individual $i$  now represents survival since the previous event  at $t_{i,j-1}$, written here in terms of the cumulative hazard from $t_{i,j-1}$ to $t_{i,j}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
S(t_{i,j} {{!}} t_{i,j-1};\psi_i) &amp;amp;=&amp;amp; \prob{T_{i,j} &amp;gt; t_{i,j}\, {{!}} \,T_{i,j-1} = t_{i,j-1};\psi_i} \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{i,j-1},t_{i,j};\psi_i)}  \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \exp\left({-\int_{t_{i,j-1} }^{t_{i,j} } \hazard(t;\psi_i) \, dt}\right) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--%In the most simple case, $y_i$ is a vector of known event times: $y_i = (t_{i1},t_{i2},\ldots,t_{i\,n_i}).$ --&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Censoring and probability distributions==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Taking into account censoring for repeated events is slightly more complicated than for one-off events.&lt;br /&gt;
First, let us assume that a trial starts at time $t_{start}$ and ends at time $t_{stop}$. Let $(T_{i1}, T_{i2}, \ldots )$ be random event times after $t_{start}$. Then, we can distinguish between the two following situations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
1. ''Exactly observed events:'' A sequence of $n_i$ event times  is precisely observed before $t_{stop}$, i.e., ${\rm y_i = (t_{i,1},t_{i,2},\ldots,t_{i,n_i}, \quad t_{i,n_i+1}&amp;gt;\tstop)}$. &lt;br /&gt;
&lt;br /&gt;
: The conditional pdf of $y_i$ is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;repeatcensor&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \left(\prod_{j=1}^{n_i}\hazard(t_{ij};\psi_i)e^{-\cumhaz(t_{i,j-1},t_{i,j};\psi_i)}  \right)e^{-\cumhaz(t_{n_i},\tstop;\psi_i)} ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
: where $t_{i0}=\tstart$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ExampleWith2Tables&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=&lt;br /&gt;
|text=&lt;br /&gt;
Suppose that for individual $i=1$ we know there were 8 events but only 7 of them occurred before $\tstop$. Here is a graphic showing the events that were exactly observed:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival4.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This data is then stored in the table on the left below. We see that the 8th and final event is noted &amp;quot;event = 0&amp;quot; with time $\tstop = 18$, indicating that the event was not observed at the end of the time period $\tstop$. In the table on the right, we show the contributions of each observation to the conditional pdf of $y_1$. Indeed, equation [[#repeatcensor|(1)]] means that the pdf of $y_1=(y_{1,1}, \ldots, y_{1,8})$ is the product of the conditional pdfs given in the right table.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
|table1=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot;  align=&amp;quot;center&amp;quot; style=&amp;quot;width:120%; margin-left:10%;margin-right:10%&amp;quot;&lt;br /&gt;
!{{!}} ID  {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 1.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3.5 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 4.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 5.6 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 9.7 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 11.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 15.8 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 18 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}}&lt;br /&gt;
&lt;br /&gt;
|table2 =&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:200%; margin-right:10%; margin-left:10%&amp;quot;&lt;br /&gt;
!{{!}} pdf &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(1.4;\psi_1)e^{-\cumhaz(0,1.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(3.5;\psi_1)e^{-\cumhaz(1.4,3.5;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(4.4;\psi_1)e^{-\cumhaz(3.5,4.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(5.6;\psi_1)e^{-\cumhaz(4.4,5.6;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(9.7;\psi_1)e^{-\cumhaz(5.6,9.7;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(11.4;\psi_1)e^{-\cumhaz(9.7,11.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(15.8;\psi_1)e^{-\cumhaz(11.4,15.8;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(15,18;\psi_1)}$&lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
2. ''Interval-censored events:'' Let $(b_{0}, b_1], (b_{1}, b_2], \ldots , (b_{K-1}, b_K]$ be a sequence of successive intervals with $\tstart=b_0&amp;lt;b_1&amp;lt;b_2 &amp;lt; \ldots &amp;lt;b_K  = \tstop$. We do not know the exact event times, but a sequence $(m_{ik}; \, 1 \leq k  \leq  K)$ is observed, where $m_{ik}$ is the number of events that occurred for individual $i$ in interval  $(b_{k-1}, b_k]$.&lt;br /&gt;
&lt;br /&gt;
: We can show that the conditional pdf of $y_i$ is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;pdf_mult_int&amp;quot; &amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) =  \prod_{k=1}^{K} e^{-\cumhaz(b_{k-1}, b_k;\psi_i)} \displaystyle{\frac{\cumhaz^{m_{ik} }(b_{k-1}, b_k;\psi_i)}{m_{ik}!} } .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
: In other words, the number of events per interval for individual $i$ is a (possibly non-homogeneous) Poisson process with intensity $\cumhaz(b_{k-1}, b_k;\psi_i)$ in interval $(b_{k-1}, b_k]$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWith2Tables&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=&lt;br /&gt;
&lt;br /&gt;
|text= Here is a graphic that shows an example of the interval boundaries and the number of events that occurred in each interval for individual $i=1$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival5.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The table on the left below shows the same data. Using [[#pdf_mult_int|(2)]] we see that the conditional pdf of $y_1=(y_{1,1}, \ldots, y_{1,6})$ is the product of the conditional pdfs given in the table on the right.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
|table1=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:120%; margin-left:10%;margin-right:10%&amp;quot;&lt;br /&gt;
!{{!}} ID  {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 6 {{!}}{{!}} 3 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 9 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 12 {{!}}{{!}} 2 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 15 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 18 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}}&lt;br /&gt;
&lt;br /&gt;
|table2=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:200%; margin-right:10%; margin-left:10% &amp;quot;&lt;br /&gt;
!{{!}} pdf &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(0,3;\psi_1)}\cumhaz(0,3;\psi_1)  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(3,6;\psi_1)} {\cumhaz^{3}(3,6;\psi_1)}/{6}  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(6,9;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(9,12;\psi_1)} {\cumhaz^{2}(9,12;\psi_1)}/{2}  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(12,15;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(15,18;\psi_1)}\cumhaz(15,18;\psi_1)  $&lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= if the total number $n_i$ of (observed and unobserved) events for individual $i$ is known to be finite, then formula [[#pdf_mult_int|(2)]] is slightly modified when the last event occurs before $\tstop$ ($t_{n_i}&amp;lt;\tstop$).&lt;br /&gt;
Assume that the last event for individual $i$ occurs in the $K_i$-th interval.  Let $s_{i} = \sum_{i=1}^{k_i-1} m_{ik}$ be the number of events that occurred before this interval. Then, we can show that&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef_Special&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;pdf_mult_int2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \prod_{k=1}^{K_i-1} \left( \displaystyle{ \frac{\cumhaz^{m_{ik} }(b_{k-1}, b_k;\psi_i)}{m_{ik}!}  }e^{-\cumhaz(b_{k-1}, b_k;\psi_i)} \right)&lt;br /&gt;
\!\times \!\left(1 - \sum_{\ell=0}^{n_i-s_{i} }   \displaystyle{ \frac{\cumhaz^{\ell}(b_{k_i -1},b_{k_i};\psi_i)}{\ell!} } e^{-\cumhaz(b_{k_i -1},b_{k_i};\psi_i)}\right) . &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Examples of hazard functions==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant hazard model:'' &lt;br /&gt;
: The most simple case is that of a constant hazard function: $\hazard(t;\psi_i) = \hazard_i \in \Rset$. Here, $\psi_i=\hazard_i$. &lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional hazards model:''&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\psi_i) = \hazard_0(t;\alpha_i) \, e^{  \langle \beta , c_i  \rangle}.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
: Here, the hazard is decomposed into two terms: a baseline function $\hazard_0$ of $t$, and an &amp;quot;individual&amp;quot; term, function of some individual covariates $c_i$. $ \langle \beta , c_i  \rangle$ means a scalar product, i.e., a linear function of $c_i$. In a proportional hazards model, a unit increase in the value of a covariate has a multiplicative effect on the hazard.&lt;br /&gt;
&lt;br /&gt;
: In the usual proportional hazard model, $\alpha_i$ is a population constant ($\alpha_i=\alpha$). Then, $\psi_i$ can be decomposed into a set of population parameters $\alpha$ and an individual parameter $ \langle \beta , c_i  \rangle$. A straightforward extension consists in assuming that $\alpha_i$ is also an individual parameter.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Extended proportional hazards model:''&lt;br /&gt;
&lt;br /&gt;
: Another possible extension assumes that the hazard function is a (possibly nonlinear) function $u$ of  a regression variable $x_i$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\bpsi_i) = \hazard_0(t;\alpha_{i}) \, e^{  u(\beta_i,x_i(t))} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Consider for example that $x_i(t)$ is the plasmatic concentration of a drug at time $t$ for individual $i$. Then, $u(\beta_i,x_i(t))$ is the term that represents (i.e., models) the effect of the drug  on the hazard, while $\hazard_0(t;\alpha_i)$  might model the effect of disease progression on the hazard.&lt;br /&gt;
&amp;lt;!--%We consider here parametric functions that possibly depend on individual parameters.--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
: In this example, $x_i(t)$ is the &amp;quot;true&amp;quot; plasmatic concentration for subject $i$ at time $t$, and it is a continuous function of time. However, in practice it is only measured at precise times, so a longitudinal model for plasmatic concentration is needed to give a concentration value for each $t$.&lt;br /&gt;
:Therefore, in practice we need to develop a ''joint model'' in order to simultaneously model time-to-events data and longitudinal data. Such an approach is introduced in the [[Joint models]] section.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
 &lt;br /&gt;
&lt;br /&gt;
* ''Accelerated failure time (AFT) model:''&lt;br /&gt;
&lt;br /&gt;
:Unlike proportional hazards models, the AFT model supposes that a change in a covariate has a multiplicative effect not on the hazard but the ''predicted event time''. This can be written as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(T_i) =  \langle \psi_i , c_i  \rangle + \xi_i&lt;br /&gt;
&amp;lt;/math&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
: where $\xi_i$ is a zero-mean random variable, e.g., a centered normal distribution. Usually, parameters are fixed effects: $\psi_i=\psi$ for each subject $i$.&lt;br /&gt;
: To calculate the hazard function, let us first denote  $p_{\xi_i}$ the density and $F_{\xi_i}$ the cdf of $\xi_i$, and to simplify, denote $\mu_i = \langle \psi_i , c_i  \rangle$ the mean of $\log(T_i)$. We begin by calculating the survival function:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
S(t;\psi_i) &amp;amp;=&amp;amp; \prob{\log{T_i} &amp;gt; \log{t} ; \bpsi_i} \\&lt;br /&gt;
&amp;amp;=&amp;amp; \int_{\log{t}-\mu_i}^{\infty} p_{\xi_i}(u; \psi_i) \, du \\&lt;br /&gt;
&amp;amp;=&amp;amp; 1 - F_{\xi_i}(\log{t}-\mu_i ; \psi_i) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Calculating [[#HazardSurvival|(1)]] then gives the hazard function:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\psi_i) = \displaystyle{ \frac{p_{\xi_i}(\log{t} - \mu_i; \psi_i)}{t(1- F_{\xi_i}(\log{t} - \mu_i; \psi_i))} }\,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
-------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary &lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=&lt;br /&gt;
For a given vector of individual parameters $\psi_i$, a model for (repeated)  time-to-event data is completely defined by&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the hazard function $\hazard(t ; \psi_i)$, or the survival function $S(t ; \psi_i)$ &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (possibly) the interval and/or right censoring process &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (possibly) the maximum number of possible events &amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--&lt;br /&gt;
==$\mlxtran$ for time-to-event data models==&lt;br /&gt;
--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{aalen2008,&lt;br /&gt;
author = {Aalen, O. and Borgan, O. and Gjessing, H.},&lt;br /&gt;
title  = {Survival and Event History Analysis. },&lt;br /&gt;
publisher = {Springer},&lt;br /&gt;
address = {New York},&lt;br /&gt;
year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{andersen2006survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Andersen, P. K.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{diggle1994,&lt;br /&gt;
author = {Diggle, P. and Kenward, M. G.},&lt;br /&gt;
title  = {Informative drop-out in longitudinal data analysis.},&lt;br /&gt;
journal = {Appl. Stats},&lt;br /&gt;
volume = {43},&lt;br /&gt;
number = {},&lt;br /&gt;
pages = {49-93},&lt;br /&gt;
year = {1994}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{duchateau2008,&lt;br /&gt;
author = {Duchateau, L. and Janssen, P.},&lt;br /&gt;
title  = {The Frailty Model. Statistics for Biology and Health },&lt;br /&gt;
publisher = {Springer.},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {2008},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fleming2011counting,&lt;br /&gt;
  title={Counting processes and survival analysis},&lt;br /&gt;
  author={Fleming, T. R. and Harrington, D. P.},&lt;br /&gt;
  volume={169},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{huang2007,&lt;br /&gt;
author = {Huang, X. and Liu, L.},&lt;br /&gt;
title  = {A joint frailty model for survival and gap times between recurrent events.},&lt;br /&gt;
journal = {Biometrics},&lt;br /&gt;
volume = {63},&lt;br /&gt;
number = {},&lt;br /&gt;
pages = {389-397},&lt;br /&gt;
year = {2007}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ibrahim2005bayesian,&lt;br /&gt;
  title={Bayesian survival analysis},&lt;br /&gt;
  author={Ibrahim, J. G. and Chen, M.-H. and Sinha, D.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{kalbfleisch2011statistical,&lt;br /&gt;
  title={The statistical analysis of failure time data},&lt;br /&gt;
  author={Kalbfleisch, J. D. and Prentice, R. L.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{kelly2000,&lt;br /&gt;
author = {Kelly, P. J. and Jim, L. L.},&lt;br /&gt;
title  = {Survival analysis for recurrent event data: an application to childhood infectious disease.},&lt;br /&gt;
journal = {Statistics in Medicine},&lt;br /&gt;
volume = {19},&lt;br /&gt;
number = {1},&lt;br /&gt;
pages = {13-33},&lt;br /&gt;
year = {2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{klein2003survival,&lt;br /&gt;
  title={Survival analysis: techniques for censored and truncated data},&lt;br /&gt;
  author={Klein, J. P. and Moeschberger, M. L.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{klein1997,&lt;br /&gt;
author = {Klein, J. P. and Moeschberger, M. L.},&lt;br /&gt;
title  = { Survival Analysis - Techniques for Censored and Truncated Data. },&lt;br /&gt;
publisher = {Springer-Verlag},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {1997},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{kleinbaum2011survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Kleinbaum, D. G.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{miller2011survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Miller Jr, R. G.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wienke2010frailty,&lt;br /&gt;
  title={Frailty models in survival analysis},&lt;br /&gt;
  author={Wienke, A.},&lt;br /&gt;
  volume={37},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Chapman &amp;amp; Hall}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Model for categorical data&lt;br /&gt;
|linkNext=Joint models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_time-to-event_data&amp;diff=7432</id>
		<title>Models for time-to-event data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_time-to-event_data&amp;diff=7432"/>
		<updated>2013-06-25T13:09:31Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Probability distributions */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, observations are the &amp;quot;times at which events occur&amp;quot;. An event may be one-off (e.g., death, hardware failure) or repeated (e.g., epileptic seizures, metro strike).&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Single event==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
To begin with, we will consider a one-off event.&lt;br /&gt;
Depending on the application, the length of time to this event  may be called the ''survival'' time (until death), ''failure'' time (until hardware fails), etc. To be general, we can just say ''event'' time.&lt;br /&gt;
&lt;br /&gt;
The random variable representing the event time for subject $i$ is typically written $T_i$. Several situations are then possible to define the observations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The event time is exactly observed.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: Then, the observation for individual $i$ is $y_i = t_i$, where $t_i$ is a realization of the random variable $T_i$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* We may know the event has happened in an interval $I_i$ but not know the exact time $t_i$. This is ''interval censoring''. For example, at a routine check-up, cancer recurrence may be detected, and we only know that it has occurred at some point in time since the last check-up.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival3.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: The observation for individual $i$ is the event: $y_i = $ &amp;quot;$a_i &amp;lt; t_i \leq b_i$&amp;quot;.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If we assume that the trial ends at time $\tstop$, then the event may happen after the end of the trial period. This is ''right censoring''.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: There are several variations of this for defining what the observations are:&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If events (before $\tstop$) are exactly observed, then for $i=1,2,\ldots, N$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1|&lt;br /&gt;
equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_i = \left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
t_i &amp;amp; {\rm if \quad} t_i \leq \tstop \\&lt;br /&gt;
{\rm t_i &amp;gt; \tstop \quad} &amp;amp; {\rm otherwise. \quad}&lt;br /&gt;
\end{array} \right.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithText&amp;amp;Table&lt;br /&gt;
|title1=Example:&lt;br /&gt;
|title2=&lt;br /&gt;
|equation=&lt;br /&gt;
Assume  that a trial starts at $\tstart=0$ and ends at $\tstop=5$, and that we obtain the following observations from 4 individuals: &lt;br /&gt;
&lt;br /&gt;
$y_1 = 3.2$ &lt;br /&gt;
&lt;br /&gt;
$y_2=$ &amp;quot;$t_2&amp;gt;5$&amp;quot;&lt;br /&gt;
&lt;br /&gt;
$y_3= 2.7$ &lt;br /&gt;
&lt;br /&gt;
$y_4 =$ &amp;quot;$t_4&amp;gt;5$&amp;quot;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These observations can be stored in a data file as shown in the table on the right.&lt;br /&gt;
&lt;br /&gt;
Here, &amp;quot;event=0&amp;quot; at time $t$ means that the event happened after $t$ while &amp;quot;event=1&amp;quot; means that the event happened at time $t$. &lt;br /&gt;
&lt;br /&gt;
The lines with $t=0$ are used to state the trial start time $\tstart=0$.&lt;br /&gt;
&lt;br /&gt;
|table=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 75%&amp;quot;&lt;br /&gt;
!{{!}} ID {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3.2 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 2.7 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}} &lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* If events before $\tstop$ are interval censored, then for $i=1,2,\ldots, N$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_i = \left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
{\rm a_i &amp;lt; t_i \quad \leq \quad b_i} &amp;amp; {\rm if \quad} t_i\leq \tstop \\&lt;br /&gt;
{\rm t_i  &amp;gt;  \tstop \quad} &amp;amp; {\rm otherwise.}&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithText&amp;amp;Table&lt;br /&gt;
|title1=Example:&lt;br /&gt;
|title2=&lt;br /&gt;
|equation=&lt;br /&gt;
Assume that we have censoring intervals of length 1: &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
$(0,1],(1,2],\ldots,(4,5]$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the same four individuals as the previous example, we now have the following observations: &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
$y_1=$  &amp;quot;$3 &amp;lt; t_1 \leq 4$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_2=$  &amp;quot;$t_2&amp;gt;5$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_3=$ &amp;quot;$2&amp;lt; t_3 \leq 3$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_4=$  &amp;quot;$t_4&amp;gt;5$&amp;quot;. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These observations can be stored in a data file as shown in the table on the right.&lt;br /&gt;
&lt;br /&gt;
Here &amp;quot;event=0&amp;quot; at time $t$ means that the event happened after $t$ while &amp;quot;event=1&amp;quot; means that the event happened before time $t$.&lt;br /&gt;
|table=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 75%&amp;quot;&lt;br /&gt;
!{{!}} ID {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 2 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 3 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Probability distributions == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Several functions play key roles in time-to-event analysis: the [http://en.wikipedia.org/wiki/Survival_function survival function] the [http://en.wikipedia.org/wiki/Survival_analysis#Hazard_function_and_cumulative_hazard_function hazard function] and the [http://en.wikipedia.org/wiki/Survival_analysis#Hazard_function_and_cumulative_hazard_function cumulative hazard function].&lt;br /&gt;
We are still working under a population approach here and so these functions, detailed below, are therefore individual functions, i.e., each subject has its own. As we are using parametric models, this means that these functions depend on individual parameters $(\psi_i)$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The '''survival function''' $S(t; \psi_i)$ gives the probability that the event happens to individual $i$ after time $t&amp;gt;t_{start}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
S(t; \psi_i) \ \ \eqdef \ \ \prob{T_i&amp;gt;t ; \psi_i} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* The '''hazard function''' $\hazard(t;\psi_i)$ is defined for individual $i$ as the instantaneous rate of the event at time $t$, given that the event has not already occurred:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;  &lt;br /&gt;
\hazard(t;\psi_i) \ \ \eqdef \ \  \lim_{dt\to 0} \displaystyle{\frac{S(t;\psi_i) - S(t + dt;\psi_i)}{ S(t;\psi_i) \, dt} }. &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: This is equivalent to: &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;HazardSurvival&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\hazard(t;\psi_i) \ \ = \ \ -\displaystyle{ \frac{d}{dt} } \log{S(t;\psi_i)}. &lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1)&lt;br /&gt;
}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Another useful quantity is the '''cumulative hazard function''' $\cumhaz(a,b;\psi_i)$,  defined for individual $i$ as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\cumhaz(a,b;\psi_i) \ \ \eqdef \ \ \displaystyle{\int_a^b \hazard(t;\psi_i) \, dt }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
: Note that [[#HazardSurvival|(1)]] implies that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
S(t;\psi_i) \ \ = \ \ e^{-\cumhaz(t_{start},t;\psi_i)}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Equation [[#HazardSurvival|(1)]] shows that the hazard function $\hazard(t;\psi_i)$ characterizes the problem, because knowing it is the same as knowing the survival function $S(t;\psi_i)$. The probability distribution of survival data is therefore completely defined by the hazard function.&lt;br /&gt;
Let $\qcyipsii$ be the conditional distribution of the  observation $y_i$ given the vector of individual parameters $\psi_i$. Its pdf can be easily computed for the various censoring situations discussed above:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is exactly observed with $y_i=t_i$, the density is the derivative of the cumulative density function, i.e., the derivative of $1 - S(t_i;\psi_i)$:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\begin{eqnarray}\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \frac{d}{dt_i}\left(1 - e^{-\cumhaz(t_{start},t_i;\psi_i)}\right)\\&lt;br /&gt;
%&amp;amp;=&amp;amp; \left(\frac{d}{dt_i} \int_{t_{start} }^{t_i} \hazard(u;\psi_i) \, du   \right)  e^{-\cumhaz(t_{start},t_i;\psi_i)}\\&lt;br /&gt;
&amp;amp;=&amp;amp;\hazard(t_i;\psi_i)e^{-\cumhaz(t_{start},t_i;\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is interval-censored with $y_i=\,$  &amp;quot;$a_i&amp;lt;t_i\leq b_i$&amp;quot;:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \prob{T_i \in (a_i,b_i]\,{{!}} \,\psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; \prob{T_i \leq b_i {{!}} \psi_i} - \prob{T_i \leq a_i {{!}} \psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; (1-S( b_i ; \psi_i)) - (1-S( a_i ; \psi_i)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{start},a_i;\psi_i)} - e^{-\cumhaz(t_{start},b_i;\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is right-censored with $y_i= \,$ &amp;quot;$t_i&amp;gt;t_{stop}$&amp;quot;:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \prob{T_i &amp;gt; t_{stop}   {{!}} \psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; S( t_{stop} ; \psi_i) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{start},t_{stop};\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Repeated events==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Sometimes, an event can potentially happen again and again, e.g., epileptic seizures, heart attacks,  etc.&lt;br /&gt;
For any given hazard function $\hazard$, the survival function $S$ for individual $i$  now represents survival since the previous event  at $t_{i,j-1}$, written here in terms of the cumulative hazard from $t_{i,j-1}$ to $t_{i,j}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
S(t_{i,j} {{!}} t_{i,j-1};\psi_i) &amp;amp;=&amp;amp; \prob{T_{i,j} &amp;gt; t_{i,j}\, {{!}} \,T_{i,j-1} = t_{i,j-1};\psi_i} \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{i,j-1},t_{i,j};\psi_i)}  \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \exp\left({-\int_{t_{i,j-1} }^{t_{i,j} } \hazard(t;\psi_i) \, dt}\right) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--%In the most simple case, $y_i$ is a vector of known event times: $y_i = (t_{i1},t_{i2},\ldots,t_{i\,n_i}).$ --&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Censoring and probability distributions==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Taking into account censoring for repeated events is slightly more complicated than for one-off events.&lt;br /&gt;
First, let us assume that a trial starts at time $t_{start}$ and ends at time $t_{stop}$. Let $(T_{i1}, T_{i2}, \ldots )$ be random event times after $t_{start}$. Then, we can distinguish between the two following situations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
1. ''Exactly observed events:'' A sequence of $n_i$ event times  is precisely observed before $t_{stop}$, i.e., ${\rm y_i = (t_{i,1},t_{i,2},\ldots,t_{i,n_i}, \quad t_{i,n_i+1}&amp;gt;\tstop)}$. &lt;br /&gt;
&lt;br /&gt;
: The conditional pdf of $y_i$ is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;repeatcensor&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \left(\prod_{j=1}^{n_i}\hazard(t_{ij};\psi_i)e^{-\cumhaz(t_{i,j-1},t_{i,j};\psi_i)}  \right)e^{-\cumhaz(t_{n_i},\tstop;\psi_i)} ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
: where $t_{i0}=\tstart$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ExampleWith2Tables&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=&lt;br /&gt;
|text=&lt;br /&gt;
Suppose that for individual $i=1$ we know there were 8 events but only 7 of them occurred before $\tstop$. Here is a graphic showing the events that were exactly observed:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival4.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This data is then stored in the table on the left below. We see that the 8th and final event is noted &amp;quot;event = 0&amp;quot; with time $\tstop = 18$, indicating that the event was not observed at the end of the time period $\tstop$. In the table on the right, we show the contributions of each observation to the conditional pdf of $y_1$. Indeed, equation [[#repeatcensor|(1)]] means that the pdf of $y_1=(y_{1,1}, \ldots, y_{1,8})$ is the product of the conditional pdfs given in the right table.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
|table1=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot;  align=&amp;quot;center&amp;quot; style=&amp;quot;width:120%; margin-left:10%;margin-right:10%&amp;quot;&lt;br /&gt;
!{{!}} ID  {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 1.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3.5 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 4.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 5.6 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 9.7 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 11.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 15.8 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 18 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}}&lt;br /&gt;
&lt;br /&gt;
|table2 =&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:200%; margin-right:10%; margin-left:10%&amp;quot;&lt;br /&gt;
!{{!}} pdf &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(1.4;\psi_1)e^{-\cumhaz(0,1.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(3.5;\psi_1)e^{-\cumhaz(1.4,3.5;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(4.4;\psi_1)e^{-\cumhaz(3.5,4.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(5.6;\psi_1)e^{-\cumhaz(4.4,5.6;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(9.7;\psi_1)e^{-\cumhaz(5.6,9.7;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(11.4;\psi_1)e^{-\cumhaz(9.7,11.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(15.8;\psi_1)e^{-\cumhaz(11.4,15.8;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(15,18;\psi_1)}$&lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
2. ''Interval-censored events:'' Let $(b_{0}, b_1], (b_{1}, b_2], \ldots , (b_{K-1}, b_K]$ be a sequence of successive intervals with $\tstart=b_0&amp;lt;b_1&amp;lt;b_2 &amp;lt; \ldots &amp;lt;b_K  = \tstop$. We do not know the exact event times, but a sequence $(m_{ik}; \, 1 \leq k  \leq  K)$ is observed, where $m_{ik}$ is the number of events that occurred for individual $i$ in interval  $(b_{k-1}, b_k]$.&lt;br /&gt;
&lt;br /&gt;
: We can show that the conditional pdf of $y_i$ is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;pdf_mult_int&amp;quot; &amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) =  \prod_{k=1}^{K} e^{-\cumhaz(b_{k-1}, b_k;\psi_i)} \displaystyle{\frac{\cumhaz^{m_{ik} }(b_{k-1}, b_k;\psi_i)}{m_{ik}!} } .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
: In other words, the number of events per interval for individual $i$ is a (possibly non-homogeneous) Poisson process with intensity $\cumhaz(b_{k-1}, b_k;\psi_i)$ in interval $(b_{k-1}, b_k]$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWith2Tables&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=&lt;br /&gt;
&lt;br /&gt;
|text= Here is a graphic that shows an example of the interval boundaries and the number of events that occurred in each interval for individual $i=1$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival5.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The table on the left below shows the same data. Using [[#pdf_mult_int|(2)]] we see that the conditional pdf of $y_1=(y_{1,1}, \ldots, y_{1,6})$ is the product of the conditional pdfs given in the table on the right.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
|table1=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:120%; margin-left:10%;margin-right:10%&amp;quot;&lt;br /&gt;
!{{!}} ID  {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 6 {{!}}{{!}} 3 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 9 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 12 {{!}}{{!}} 2 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 15 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 18 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}}&lt;br /&gt;
&lt;br /&gt;
|table2=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:200%; margin-right:10%; margin-left:10% &amp;quot;&lt;br /&gt;
!{{!}} pdf &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(0,3;\psi_1)}\cumhaz(0,3;\psi_1)  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(3,6;\psi_1)} {\cumhaz^{3}(3,6;\psi_1)}/{6}  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(6,9;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(9,12;\psi_1)} {\cumhaz^{2}(9,12;\psi_1)}/{2}  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(12,15;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(15,18;\psi_1)}\cumhaz(15,18;\psi_1)  $&lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= if the total number $n_i$ of (observed and unobserved) events for individual $i$ is known to be finite, then formula [[#pdf_mult_int|(2)]] is slightly modified when the last event occurs before $\tstop$ ($t_{n_i}&amp;lt;\tstop$).&lt;br /&gt;
Assume that the last event for individual $i$ occurs in the $K_i$-th interval.  Let $s_{i} = \sum_{i=1}^{k_i-1} m_{ik}$ be the number of events that occurred before this interval. Then, we can show that&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef_Special&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;pdf_mult_int2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \prod_{k=1}^{K_i-1} \left( \displaystyle{ \frac{\cumhaz^{m_{ik} }(b_{k-1}, b_k;\psi_i)}{m_{ik}!}  }e^{-\cumhaz(b_{k-1}, b_k;\psi_i)} \right)&lt;br /&gt;
\!\times \!\left(1 - \sum_{\ell=0}^{n_i-s_{i} }   \displaystyle{ \frac{\cumhaz^{\ell}(b_{k_i -1},b_{k_i};\psi_i)}{\ell!} } e^{-\cumhaz(b_{k_i -1},b_{k_i};\psi_i)}\right) . &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Examples of hazard functions==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant hazard model:'' &lt;br /&gt;
: The most simple case is that of a constant hazard function: $\hazard(t;\psi_i) = \hazard_i \in \Rset$. Here, $\psi_i=\hazard_i$. &lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional hazards model:''&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\psi_i) = \hazard_0(t;\alpha_i) \, e^{  \langle \beta , c_i  \rangle}.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
: Here, the hazard is decomposed into two terms: a baseline function $\hazard_0$ of $t$, and an &amp;quot;individual&amp;quot; term, function of some individual covariates $c_i$. $ \langle \beta , c_i  \rangle$ means a scalar product, i.e., a linear function of $c_i$. In a proportional hazards model, a unit increase in the value of a covariate has a multiplicative effect on the hazard.&lt;br /&gt;
&lt;br /&gt;
: In the usual proportional hazard model, $\alpha_i$ is a population constant ($\alpha_i=\alpha$). Then, $\psi_i$ can be decomposed into a set of population parameters $\alpha$ and an individual parameter $ \langle \beta , c_i  \rangle$. A straightforward extension consists in assuming that $\alpha_i$ is also an individual parameter.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Extended proportional hazards model:''&lt;br /&gt;
&lt;br /&gt;
: Another possible extension assumes that the hazard function is a (possibly nonlinear) function $u$ of  a regression variable $x_i$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\bpsi_i) = \hazard_0(t;\alpha_{i}) \, e^{  u(\beta_i,x_i(t))} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Consider for example that $x_i(t)$ is the plasmatic concentration of a drug at time $t$ for individual $i$. Then, $u(\beta_i,x_i(t))$ is the term that represents (i.e., models) the effect of the drug  on the hazard, while $\hazard_0(t;\alpha_i)$  might model the effect of disease progression on the hazard.&lt;br /&gt;
&amp;lt;!--%We consider here parametric functions that possibly depend on individual parameters.--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
: In this example, $x_i(t)$ is the &amp;quot;true&amp;quot; plasmatic concentration for subject $i$ at time $t$, and it is a continuous function of time. However, in practice it is only measured at precise times, so a longitudinal model for plasmatic concentration is needed to give a concentration value for each $t$.&lt;br /&gt;
:Therefore, in practice we need to develop a ''joint model'' in order to simultaneously model time-to-events data and longitudinal data. Such an approach is introduced in the [[Joint models]] section.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
 &lt;br /&gt;
&lt;br /&gt;
* ''Accelerated failure time (AFT) model:''&lt;br /&gt;
&lt;br /&gt;
:Unlike proportional hazards models, the AFT model supposes that a change in a covariate has a multiplicative effect not on the hazard but the ''predicted event time''. This can be written as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(T_i) =  \langle \psi_i , c_i  \rangle + \xi_i&lt;br /&gt;
&amp;lt;/math&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
: where $\xi_i$ is a zero-mean random variable, e.g., a centered normal distribution. Usually, parameters are fixed effects: $\psi_i=\psi$ for each subject $i$.&lt;br /&gt;
: To calculate the hazard function, let us first denote  $p_{\xi_i}$ the density and $F_{\xi_i}$ the cdf of $\xi_i$, and to simplify, denote $\mu_i = \langle \psi_i , c_i  \rangle$ the mean of $\log(T_i)$. We begin by calculating the survival function:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
S(t;\psi_i) &amp;amp;=&amp;amp; \prob{\log{T_i} &amp;gt; \log{t} ; \bpsi_i} \\&lt;br /&gt;
&amp;amp;=&amp;amp; \int_{\log{t}-\mu_i}^{\infty} p_{\xi_i}(u; \psi_i) \, du \\&lt;br /&gt;
&amp;amp;=&amp;amp; 1 - F_{\xi_i}(\log{t}-\mu_i ; \psi_i) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Calculating [[#HazardSurvival|(1)]] then gives the hazard function:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\psi_i) = \displaystyle{ \frac{p_{\xi_i}(\log{t} - \mu_i; \psi_i)}{t(1- F_{\xi_i}(\log{t} - \mu_i; \psi_i))} }\,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
-------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary &lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=&lt;br /&gt;
For a given vector of individual parameters $\psi_i$, a model for (repeated)  time-to-event data is completely defined by&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the hazard function $\hazard(t ; \psi_i)$, or the survival function $S(t ; \psi_i)$ &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (possibly) the interval and/or right censoring process &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (possibly) the maximum number of possible events &amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--&lt;br /&gt;
==$\mlxtran$ for time-to-event data models==&lt;br /&gt;
--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{aalen2008,&lt;br /&gt;
author = {Aalen, O. and Borgan, O. and Gjessing, H.},&lt;br /&gt;
title  = {Survival and Event History Analysis. },&lt;br /&gt;
publisher = {Springer},&lt;br /&gt;
address = {New York},&lt;br /&gt;
year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{andersen2006survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Andersen, P. K.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{diggle1994,&lt;br /&gt;
author = {Diggle, P. and Kenward, M. G.},&lt;br /&gt;
title  = {Informative drop-out in longitudinal data analysis.},&lt;br /&gt;
journal = {Appl. Stats},&lt;br /&gt;
volume = {43},&lt;br /&gt;
number = {},&lt;br /&gt;
pages = {49-93},&lt;br /&gt;
year = {1994}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{duchateau2008,&lt;br /&gt;
author = {Duchateau, L. and Janssen, P.},&lt;br /&gt;
title  = {The Frailty Model. Statistics for Biology and Health },&lt;br /&gt;
publisher = {Springer.},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {2008},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fleming2011counting,&lt;br /&gt;
  title={Counting processes and survival analysis},&lt;br /&gt;
  author={Fleming, T. R. and Harrington, D. P.},&lt;br /&gt;
  volume={169},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{huang2007,&lt;br /&gt;
author = {Huang, X. and Liu, L.},&lt;br /&gt;
title  = {A joint frailty model for survival and gap times between recurrent events.},&lt;br /&gt;
journal = {Biometrics},&lt;br /&gt;
volume = {63},&lt;br /&gt;
number = {},&lt;br /&gt;
pages = {389-397},&lt;br /&gt;
year = {2007}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ibrahim2005bayesian,&lt;br /&gt;
  title={Bayesian survival analysis},&lt;br /&gt;
  author={Ibrahim, J. G. and Chen, M.-H. and Sinha, D.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{kalbfleisch2011statistical,&lt;br /&gt;
  title={The statistical analysis of failure time data},&lt;br /&gt;
  author={Kalbfleisch, J. D. and Prentice, R. L.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{kelly2000,&lt;br /&gt;
author = {Kelly, P. J. and Jim, L. L.},&lt;br /&gt;
title  = {Survival analysis for recurrent event data: an application to childhood infectious disease.},&lt;br /&gt;
journal = {Statistics in Medicine},&lt;br /&gt;
volume = {19},&lt;br /&gt;
number = {1},&lt;br /&gt;
pages = {13-33},&lt;br /&gt;
year = {2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{klein2003survival,&lt;br /&gt;
  title={Survival analysis: techniques for censored and truncated data},&lt;br /&gt;
  author={Klein, J. P. and Moeschberger, M. L.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{klein1997,&lt;br /&gt;
author = {Klein, J. P. and Moeschberger, M. L.},&lt;br /&gt;
title  = { Survival Analysis - Techniques for Censored and Truncated Data. },&lt;br /&gt;
publisher = {Springer-Verlag},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {1997},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{kleinbaum2011survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Kleinbaum, D. G.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{miller2011survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Miller Jr, R. G.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wienke2010frailty,&lt;br /&gt;
  title={Frailty models in survival analysis},&lt;br /&gt;
  author={Wienke, A.},&lt;br /&gt;
  volume={37},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Chapman &amp;amp; Hall}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Model for categorical data&lt;br /&gt;
|linkNext=Joint models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_time-to-event_data&amp;diff=7431</id>
		<title>Models for time-to-event data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_time-to-event_data&amp;diff=7431"/>
		<updated>2013-06-25T13:08:56Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Probability distributions */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Here, observations are the &amp;quot;times at which events occur&amp;quot;. An event may be one-off (e.g., death, hardware failure) or repeated (e.g., epileptic seizures, metro strike).&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Single event==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
To begin with, we will consider a one-off event.&lt;br /&gt;
Depending on the application, the length of time to this event  may be called the ''survival'' time (until death), ''failure'' time (until hardware fails), etc. To be general, we can just say ''event'' time.&lt;br /&gt;
&lt;br /&gt;
The random variable representing the event time for subject $i$ is typically written $T_i$. Several situations are then possible to define the observations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The event time is exactly observed.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: Then, the observation for individual $i$ is $y_i = t_i$, where $t_i$ is a realization of the random variable $T_i$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* We may know the event has happened in an interval $I_i$ but not know the exact time $t_i$. This is ''interval censoring''. For example, at a routine check-up, cancer recurrence may be detected, and we only know that it has occurred at some point in time since the last check-up.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival3.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: The observation for individual $i$ is the event: $y_i = $ &amp;quot;$a_i &amp;lt; t_i \leq b_i$&amp;quot;.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If we assume that the trial ends at time $\tstop$, then the event may happen after the end of the trial period. This is ''right censoring''.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
: There are several variations of this for defining what the observations are:&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If events (before $\tstop$) are exactly observed, then for $i=1,2,\ldots, N$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1|&lt;br /&gt;
equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_i = \left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
t_i &amp;amp; {\rm if \quad} t_i \leq \tstop \\&lt;br /&gt;
{\rm t_i &amp;gt; \tstop \quad} &amp;amp; {\rm otherwise. \quad}&lt;br /&gt;
\end{array} \right.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithText&amp;amp;Table&lt;br /&gt;
|title1=Example:&lt;br /&gt;
|title2=&lt;br /&gt;
|equation=&lt;br /&gt;
Assume  that a trial starts at $\tstart=0$ and ends at $\tstop=5$, and that we obtain the following observations from 4 individuals: &lt;br /&gt;
&lt;br /&gt;
$y_1 = 3.2$ &lt;br /&gt;
&lt;br /&gt;
$y_2=$ &amp;quot;$t_2&amp;gt;5$&amp;quot;&lt;br /&gt;
&lt;br /&gt;
$y_3= 2.7$ &lt;br /&gt;
&lt;br /&gt;
$y_4 =$ &amp;quot;$t_4&amp;gt;5$&amp;quot;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These observations can be stored in a data file as shown in the table on the right.&lt;br /&gt;
&lt;br /&gt;
Here, &amp;quot;event=0&amp;quot; at time $t$ means that the event happened after $t$ while &amp;quot;event=1&amp;quot; means that the event happened at time $t$. &lt;br /&gt;
&lt;br /&gt;
The lines with $t=0$ are used to state the trial start time $\tstart=0$.&lt;br /&gt;
&lt;br /&gt;
|table=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 75%&amp;quot;&lt;br /&gt;
!{{!}} ID {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3.2 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 2.7 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}} &lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* If events before $\tstop$ are interval censored, then for $i=1,2,\ldots, N$,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
y_i = \left\{&lt;br /&gt;
\begin{array}{ll}&lt;br /&gt;
{\rm a_i &amp;lt; t_i \quad \leq \quad b_i} &amp;amp; {\rm if \quad} t_i\leq \tstop \\&lt;br /&gt;
{\rm t_i  &amp;gt;  \tstop \quad} &amp;amp; {\rm otherwise.}&lt;br /&gt;
\end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithText&amp;amp;Table&lt;br /&gt;
|title1=Example:&lt;br /&gt;
|title2=&lt;br /&gt;
|equation=&lt;br /&gt;
Assume that we have censoring intervals of length 1: &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
$(0,1],(1,2],\ldots,(4,5]$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the same four individuals as the previous example, we now have the following observations: &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
$y_1=$  &amp;quot;$3 &amp;lt; t_1 \leq 4$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_2=$  &amp;quot;$t_2&amp;gt;5$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_3=$ &amp;quot;$2&amp;lt; t_3 \leq 3$&amp;quot;, &lt;br /&gt;
&lt;br /&gt;
$y_4=$  &amp;quot;$t_4&amp;gt;5$&amp;quot;. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
These observations can be stored in a data file as shown in the table on the right.&lt;br /&gt;
&lt;br /&gt;
Here &amp;quot;event=0&amp;quot; at time $t$ means that the event happened after $t$ while &amp;quot;event=1&amp;quot; means that the event happened before time $t$.&lt;br /&gt;
|table=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width: 75%&amp;quot;&lt;br /&gt;
!{{!}} ID {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}2 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 2 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}3 {{!}}{{!}} 3 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}4 {{!}}{{!}} 5 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Probability distributions == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Several functions play key roles in time-to-event analysis: the [http://en.wikipedia.org/wiki/Survival_function survival function] the [http://en.wikipedia.org/wiki/Hazard_function#hazard_function hazard function] and the [http://en.wikipedia.org/wiki/Survival_analysis#Hazard_function_and_cumulative_hazard_function cumulative hazard function].&lt;br /&gt;
We are still working under a population approach here and so these functions, detailed below, are therefore individual functions, i.e., each subject has its own. As we are using parametric models, this means that these functions depend on individual parameters $(\psi_i)$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The '''survival function''' $S(t; \psi_i)$ gives the probability that the event happens to individual $i$ after time $t&amp;gt;t_{start}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
S(t; \psi_i) \ \ \eqdef \ \ \prob{T_i&amp;gt;t ; \psi_i} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* The '''hazard function''' $\hazard(t;\psi_i)$ is defined for individual $i$ as the instantaneous rate of the event at time $t$, given that the event has not already occurred:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;  &lt;br /&gt;
\hazard(t;\psi_i) \ \ \eqdef \ \  \lim_{dt\to 0} \displaystyle{\frac{S(t;\psi_i) - S(t + dt;\psi_i)}{ S(t;\psi_i) \, dt} }. &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: This is equivalent to: &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;HazardSurvival&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\hazard(t;\psi_i) \ \ = \ \ -\displaystyle{ \frac{d}{dt} } \log{S(t;\psi_i)}. &lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1)&lt;br /&gt;
}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Another useful quantity is the '''cumulative hazard function''' $\cumhaz(a,b;\psi_i)$,  defined for individual $i$ as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\cumhaz(a,b;\psi_i) \ \ \eqdef \ \ \displaystyle{\int_a^b \hazard(t;\psi_i) \, dt }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
: Note that [[#HazardSurvival|(1)]] implies that:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
S(t;\psi_i) \ \ = \ \ e^{-\cumhaz(t_{start},t;\psi_i)}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Equation [[#HazardSurvival|(1)]] shows that the hazard function $\hazard(t;\psi_i)$ characterizes the problem, because knowing it is the same as knowing the survival function $S(t;\psi_i)$. The probability distribution of survival data is therefore completely defined by the hazard function.&lt;br /&gt;
Let $\qcyipsii$ be the conditional distribution of the  observation $y_i$ given the vector of individual parameters $\psi_i$. Its pdf can be easily computed for the various censoring situations discussed above:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is exactly observed with $y_i=t_i$, the density is the derivative of the cumulative density function, i.e., the derivative of $1 - S(t_i;\psi_i)$:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\begin{eqnarray}\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \frac{d}{dt_i}\left(1 - e^{-\cumhaz(t_{start},t_i;\psi_i)}\right)\\&lt;br /&gt;
%&amp;amp;=&amp;amp; \left(\frac{d}{dt_i} \int_{t_{start} }^{t_i} \hazard(u;\psi_i) \, du   \right)  e^{-\cumhaz(t_{start},t_i;\psi_i)}\\&lt;br /&gt;
&amp;amp;=&amp;amp;\hazard(t_i;\psi_i)e^{-\cumhaz(t_{start},t_i;\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is interval-censored with $y_i=\,$  &amp;quot;$a_i&amp;lt;t_i\leq b_i$&amp;quot;:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \prob{T_i \in (a_i,b_i]\,{{!}} \,\psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; \prob{T_i \leq b_i {{!}} \psi_i} - \prob{T_i \leq a_i {{!}} \psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; (1-S( b_i ; \psi_i)) - (1-S( a_i ; \psi_i)) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{start},a_i;\psi_i)} - e^{-\cumhaz(t_{start},b_i;\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt;If the event is right-censored with $y_i= \,$ &amp;quot;$t_i&amp;gt;t_{stop}$&amp;quot;:&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) &amp;amp;=&amp;amp; \prob{T_i &amp;gt; t_{stop}   {{!}} \psi_i}  \\&lt;br /&gt;
%&amp;amp;=&amp;amp; S( t_{stop} ; \psi_i) \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{start},t_{stop};\psi_i)} .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Repeated events==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Sometimes, an event can potentially happen again and again, e.g., epileptic seizures, heart attacks,  etc.&lt;br /&gt;
For any given hazard function $\hazard$, the survival function $S$ for individual $i$  now represents survival since the previous event  at $t_{i,j-1}$, written here in terms of the cumulative hazard from $t_{i,j-1}$ to $t_{i,j}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
S(t_{i,j} {{!}} t_{i,j-1};\psi_i) &amp;amp;=&amp;amp; \prob{T_{i,j} &amp;gt; t_{i,j}\, {{!}} \,T_{i,j-1} = t_{i,j-1};\psi_i} \\&lt;br /&gt;
&amp;amp;=&amp;amp;  e^{-\cumhaz(t_{i,j-1},t_{i,j};\psi_i)}  \\&lt;br /&gt;
&amp;amp;=&amp;amp;  \exp\left({-\int_{t_{i,j-1} }^{t_{i,j} } \hazard(t;\psi_i) \, dt}\right) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--%In the most simple case, $y_i$ is a vector of known event times: $y_i = (t_{i1},t_{i2},\ldots,t_{i\,n_i}).$ --&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
==Censoring and probability distributions==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Taking into account censoring for repeated events is slightly more complicated than for one-off events.&lt;br /&gt;
First, let us assume that a trial starts at time $t_{start}$ and ends at time $t_{stop}$. Let $(T_{i1}, T_{i2}, \ldots )$ be random event times after $t_{start}$. Then, we can distinguish between the two following situations:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
1. ''Exactly observed events:'' A sequence of $n_i$ event times  is precisely observed before $t_{stop}$, i.e., ${\rm y_i = (t_{i,1},t_{i,2},\ldots,t_{i,n_i}, \quad t_{i,n_i+1}&amp;gt;\tstop)}$. &lt;br /&gt;
&lt;br /&gt;
: The conditional pdf of $y_i$ is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;repeatcensor&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \left(\prod_{j=1}^{n_i}\hazard(t_{ij};\psi_i)e^{-\cumhaz(t_{i,j-1},t_{i,j};\psi_i)}  \right)e^{-\cumhaz(t_{n_i},\tstop;\psi_i)} ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
: where $t_{i0}=\tstart$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{ExampleWith2Tables&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=&lt;br /&gt;
|text=&lt;br /&gt;
Suppose that for individual $i=1$ we know there were 8 events but only 7 of them occurred before $\tstop$. Here is a graphic showing the events that were exactly observed:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival4.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
This data is then stored in the table on the left below. We see that the 8th and final event is noted &amp;quot;event = 0&amp;quot; with time $\tstop = 18$, indicating that the event was not observed at the end of the time period $\tstop$. In the table on the right, we show the contributions of each observation to the conditional pdf of $y_1$. Indeed, equation [[#repeatcensor|(1)]] means that the pdf of $y_1=(y_{1,1}, \ldots, y_{1,8})$ is the product of the conditional pdfs given in the right table.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
|table1=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot;  align=&amp;quot;center&amp;quot; style=&amp;quot;width:120%; margin-left:10%;margin-right:10%&amp;quot;&lt;br /&gt;
!{{!}} ID  {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 1.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3.5 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 4.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 5.6 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 9.7 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 11.4 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 15.8 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 18 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}}&lt;br /&gt;
&lt;br /&gt;
|table2 =&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:200%; margin-right:10%; margin-left:10%&amp;quot;&lt;br /&gt;
!{{!}} pdf &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(1.4;\psi_1)e^{-\cumhaz(0,1.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(3.5;\psi_1)e^{-\cumhaz(1.4,3.5;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(4.4;\psi_1)e^{-\cumhaz(3.5,4.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(5.6;\psi_1)e^{-\cumhaz(4.4,5.6;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(9.7;\psi_1)e^{-\cumhaz(5.6,9.7;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(11.4;\psi_1)e^{-\cumhaz(9.7,11.4;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $\hazard(15.8;\psi_1)e^{-\cumhaz(11.4,15.8;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(15,18;\psi_1)}$&lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
2. ''Interval-censored events:'' Let $(b_{0}, b_1], (b_{1}, b_2], \ldots , (b_{K-1}, b_K]$ be a sequence of successive intervals with $\tstart=b_0&amp;lt;b_1&amp;lt;b_2 &amp;lt; \ldots &amp;lt;b_K  = \tstop$. We do not know the exact event times, but a sequence $(m_{ik}; \, 1 \leq k  \leq  K)$ is observed, where $m_{ik}$ is the number of events that occurred for individual $i$ in interval  $(b_{k-1}, b_k]$.&lt;br /&gt;
&lt;br /&gt;
: We can show that the conditional pdf of $y_i$ is given by:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;pdf_mult_int&amp;quot; &amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) =  \prod_{k=1}^{K} e^{-\cumhaz(b_{k-1}, b_k;\psi_i)} \displaystyle{\frac{\cumhaz^{m_{ik} }(b_{k-1}, b_k;\psi_i)}{m_{ik}!} } .&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
: In other words, the number of events per interval for individual $i$ is a (possibly non-homogeneous) Poisson process with intensity $\cumhaz(b_{k-1}, b_k;\psi_i)$ in interval $(b_{k-1}, b_k]$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWith2Tables&lt;br /&gt;
|title1=Example&lt;br /&gt;
|title2=&lt;br /&gt;
&lt;br /&gt;
|text= Here is a graphic that shows an example of the interval boundaries and the number of events that occurred in each interval for individual $i=1$.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:survival5.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The table on the left below shows the same data. Using [[#pdf_mult_int|(2)]] we see that the conditional pdf of $y_1=(y_{1,1}, \ldots, y_{1,6})$ is the product of the conditional pdfs given in the table on the right.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
|table1=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:120%; margin-left:10%;margin-right:10%&amp;quot;&lt;br /&gt;
!{{!}} ID  {{!}}{{!}} TIME  {{!}}{{!}} EVENT  &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 0 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 3 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 6 {{!}}{{!}} 3 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 9 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 12 {{!}}{{!}} 2 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 15 {{!}}{{!}} 0 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}}1 {{!}}{{!}} 18 {{!}}{{!}} 1 &lt;br /&gt;
{{!}}}&lt;br /&gt;
&lt;br /&gt;
|table2=&lt;br /&gt;
{{{!}} class=&amp;quot;wikitable&amp;quot; align=&amp;quot;center&amp;quot; style=&amp;quot;width:200%; margin-right:10%; margin-left:10% &amp;quot;&lt;br /&gt;
!{{!}} pdf &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} 1 &lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(0,3;\psi_1)}\cumhaz(0,3;\psi_1)  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(3,6;\psi_1)} {\cumhaz^{3}(3,6;\psi_1)}/{6}  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(6,9;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(9,12;\psi_1)} {\cumhaz^{2}(9,12;\psi_1)}/{2}  $&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(12,15;\psi_1)}$&lt;br /&gt;
{{!}}-&lt;br /&gt;
{{!}} $e^{-\cumhaz(15,18;\psi_1)}\cumhaz(15,18;\psi_1)  $&lt;br /&gt;
{{!}}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remark&lt;br /&gt;
|text= if the total number $n_i$ of (observed and unobserved) events for individual $i$ is known to be finite, then formula [[#pdf_mult_int|(2)]] is slightly modified when the last event occurs before $\tstop$ ($t_{n_i}&amp;lt;\tstop$).&lt;br /&gt;
Assume that the last event for individual $i$ occurs in the $K_i$-th interval.  Let $s_{i} = \sum_{i=1}^{k_i-1} m_{ik}$ be the number of events that occurred before this interval. Then, we can show that&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef_Special&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;pdf_mult_int2&amp;quot;&amp;gt;&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \prod_{k=1}^{K_i-1} \left( \displaystyle{ \frac{\cumhaz^{m_{ik} }(b_{k-1}, b_k;\psi_i)}{m_{ik}!}  }e^{-\cumhaz(b_{k-1}, b_k;\psi_i)} \right)&lt;br /&gt;
\!\times \!\left(1 - \sum_{\ell=0}^{n_i-s_{i} }   \displaystyle{ \frac{\cumhaz^{\ell}(b_{k_i -1},b_{k_i};\psi_i)}{\ell!} } e^{-\cumhaz(b_{k_i -1},b_{k_i};\psi_i)}\right) . &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Examples of hazard functions==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Constant hazard model:'' &lt;br /&gt;
: The most simple case is that of a constant hazard function: $\hazard(t;\psi_i) = \hazard_i \in \Rset$. Here, $\psi_i=\hazard_i$. &lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Proportional hazards model:''&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\psi_i) = \hazard_0(t;\alpha_i) \, e^{  \langle \beta , c_i  \rangle}.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
: Here, the hazard is decomposed into two terms: a baseline function $\hazard_0$ of $t$, and an &amp;quot;individual&amp;quot; term, function of some individual covariates $c_i$. $ \langle \beta , c_i  \rangle$ means a scalar product, i.e., a linear function of $c_i$. In a proportional hazards model, a unit increase in the value of a covariate has a multiplicative effect on the hazard.&lt;br /&gt;
&lt;br /&gt;
: In the usual proportional hazard model, $\alpha_i$ is a population constant ($\alpha_i=\alpha$). Then, $\psi_i$ can be decomposed into a set of population parameters $\alpha$ and an individual parameter $ \langle \beta , c_i  \rangle$. A straightforward extension consists in assuming that $\alpha_i$ is also an individual parameter.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* ''Extended proportional hazards model:''&lt;br /&gt;
&lt;br /&gt;
: Another possible extension assumes that the hazard function is a (possibly nonlinear) function $u$ of  a regression variable $x_i$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\bpsi_i) = \hazard_0(t;\alpha_{i}) \, e^{  u(\beta_i,x_i(t))} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Consider for example that $x_i(t)$ is the plasmatic concentration of a drug at time $t$ for individual $i$. Then, $u(\beta_i,x_i(t))$ is the term that represents (i.e., models) the effect of the drug  on the hazard, while $\hazard_0(t;\alpha_i)$  might model the effect of disease progression on the hazard.&lt;br /&gt;
&amp;lt;!--%We consider here parametric functions that possibly depend on individual parameters.--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
: In this example, $x_i(t)$ is the &amp;quot;true&amp;quot; plasmatic concentration for subject $i$ at time $t$, and it is a continuous function of time. However, in practice it is only measured at precise times, so a longitudinal model for plasmatic concentration is needed to give a concentration value for each $t$.&lt;br /&gt;
:Therefore, in practice we need to develop a ''joint model'' in order to simultaneously model time-to-events data and longitudinal data. Such an approach is introduced in the [[Joint models]] section.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
 &lt;br /&gt;
&lt;br /&gt;
* ''Accelerated failure time (AFT) model:''&lt;br /&gt;
&lt;br /&gt;
:Unlike proportional hazards models, the AFT model supposes that a change in a covariate has a multiplicative effect not on the hazard but the ''predicted event time''. This can be written as:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\log(T_i) =  \langle \psi_i , c_i  \rangle + \xi_i&lt;br /&gt;
&amp;lt;/math&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
: where $\xi_i$ is a zero-mean random variable, e.g., a centered normal distribution. Usually, parameters are fixed effects: $\psi_i=\psi$ for each subject $i$.&lt;br /&gt;
: To calculate the hazard function, let us first denote  $p_{\xi_i}$ the density and $F_{\xi_i}$ the cdf of $\xi_i$, and to simplify, denote $\mu_i = \langle \psi_i , c_i  \rangle$ the mean of $\log(T_i)$. We begin by calculating the survival function:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
S(t;\psi_i) &amp;amp;=&amp;amp; \prob{\log{T_i} &amp;gt; \log{t} ; \bpsi_i} \\&lt;br /&gt;
&amp;amp;=&amp;amp; \int_{\log{t}-\mu_i}^{\infty} p_{\xi_i}(u; \psi_i) \, du \\&lt;br /&gt;
&amp;amp;=&amp;amp; 1 - F_{\xi_i}(\log{t}-\mu_i ; \psi_i) .&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:Calculating [[#HazardSurvival|(1)]] then gives the hazard function:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\hazard(t;\psi_i) = \displaystyle{ \frac{p_{\xi_i}(\log{t} - \mu_i; \psi_i)}{t(1- F_{\xi_i}(\log{t} - \mu_i; \psi_i))} }\,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
-------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary &lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=&lt;br /&gt;
For a given vector of individual parameters $\psi_i$, a model for (repeated)  time-to-event data is completely defined by&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the hazard function $\hazard(t ; \psi_i)$, or the survival function $S(t ; \psi_i)$ &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (possibly) the interval and/or right censoring process &amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (possibly) the maximum number of possible events &amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;!--&lt;br /&gt;
==$\mlxtran$ for time-to-event data models==&lt;br /&gt;
--&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
==Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{aalen2008,&lt;br /&gt;
author = {Aalen, O. and Borgan, O. and Gjessing, H.},&lt;br /&gt;
title  = {Survival and Event History Analysis. },&lt;br /&gt;
publisher = {Springer},&lt;br /&gt;
address = {New York},&lt;br /&gt;
year = {2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{andersen2006survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Andersen, P. K.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{diggle1994,&lt;br /&gt;
author = {Diggle, P. and Kenward, M. G.},&lt;br /&gt;
title  = {Informative drop-out in longitudinal data analysis.},&lt;br /&gt;
journal = {Appl. Stats},&lt;br /&gt;
volume = {43},&lt;br /&gt;
number = {},&lt;br /&gt;
pages = {49-93},&lt;br /&gt;
year = {1994}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{duchateau2008,&lt;br /&gt;
author = {Duchateau, L. and Janssen, P.},&lt;br /&gt;
title  = {The Frailty Model. Statistics for Biology and Health },&lt;br /&gt;
publisher = {Springer.},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {2008},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fleming2011counting,&lt;br /&gt;
  title={Counting processes and survival analysis},&lt;br /&gt;
  author={Fleming, T. R. and Harrington, D. P.},&lt;br /&gt;
  volume={169},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{huang2007,&lt;br /&gt;
author = {Huang, X. and Liu, L.},&lt;br /&gt;
title  = {A joint frailty model for survival and gap times between recurrent events.},&lt;br /&gt;
journal = {Biometrics},&lt;br /&gt;
volume = {63},&lt;br /&gt;
number = {},&lt;br /&gt;
pages = {389-397},&lt;br /&gt;
year = {2007}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{ibrahim2005bayesian,&lt;br /&gt;
  title={Bayesian survival analysis},&lt;br /&gt;
  author={Ibrahim, J. G. and Chen, M.-H. and Sinha, D.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{kalbfleisch2011statistical,&lt;br /&gt;
  title={The statistical analysis of failure time data},&lt;br /&gt;
  author={Kalbfleisch, J. D. and Prentice, R. L.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{kelly2000,&lt;br /&gt;
author = {Kelly, P. J. and Jim, L. L.},&lt;br /&gt;
title  = {Survival analysis for recurrent event data: an application to childhood infectious disease.},&lt;br /&gt;
journal = {Statistics in Medicine},&lt;br /&gt;
volume = {19},&lt;br /&gt;
number = {1},&lt;br /&gt;
pages = {13-33},&lt;br /&gt;
year = {2000}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{klein2003survival,&lt;br /&gt;
  title={Survival analysis: techniques for censored and truncated data},&lt;br /&gt;
  author={Klein, J. P. and Moeschberger, M. L.},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{klein1997,&lt;br /&gt;
author = {Klein, J. P. and Moeschberger, M. L.},&lt;br /&gt;
title  = { Survival Analysis - Techniques for Censored and Truncated Data. },&lt;br /&gt;
publisher = {Springer-Verlag},&lt;br /&gt;
volume = {},&lt;br /&gt;
pages = {},&lt;br /&gt;
year = {1997},&lt;br /&gt;
series = {},&lt;br /&gt;
address = {New York},&lt;br /&gt;
edition = {},&lt;br /&gt;
month = {}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{kleinbaum2011survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Kleinbaum, D. G.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{miller2011survival,&lt;br /&gt;
  title={Survival analysis},&lt;br /&gt;
  author={Miller Jr, R. G.},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{wienke2010frailty,&lt;br /&gt;
  title={Frailty models in survival analysis},&lt;br /&gt;
  author={Wienke, A.},&lt;br /&gt;
  volume={37},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Chapman &amp;amp; Hall}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Model for categorical data&lt;br /&gt;
|linkNext=Joint models }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7430</id>
		<title>Model for categorical data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7430"/>
		<updated>2013-06-25T13:05:45Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Markovian dependence */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
== Overview == &lt;br /&gt;
&lt;br /&gt;
Assume now that the observed data  takes its values in a fixed and finite set of nominal categories $\{c_1, c_2,\ldots , c_K\}$.&lt;br /&gt;
Considering the observations $(y_{ij}, 1 \leq j \leq n_i)$ of any individual $i$ as a sequence of independent random variables, the model is completely defined by the probability mass functions $\prob{y_{ij}=c_k | \psi_i}$, for $k=1,\ldots, K$ and $1 \leq j \leq n_i$.&lt;br /&gt;
&lt;br /&gt;
For a given $(i,j)$, the sum of the $K$ probabilities is 1, so in fact only $K-1$ of them need to be defined.&lt;br /&gt;
&lt;br /&gt;
In the most general way possible, any model can be considered so long as it defines a probability distribution, i.e., for each $k$,  $\prob{y_{ij}=c_k | \psi_i} \in [0,1]$, and $\sum_{k=1}^{K} \prob{y_{ij}=c_k | \psi_i} = 1$. For instance, we could define $K$ time-dependent parametric functions $a_1$, $a_2$, ..., $a_K$  and set for any individual $i$, time $t_{ij}$ and $k \in \{1,\ldots,K\}$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;categorical1&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\prob{y_{ij}=c_k {{!}} \psi_i} = \displaystyle{\frac{e^{a_k(t_{ij},\psi_i)} }{\sum_{m=1}^K e^{a_m(t_{ij},\psi_i)} } }.   &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= Suppose we want to model binary data, i.e., data where $y_{ij} \in \{0,1\}$.&lt;br /&gt;
&lt;br /&gt;
Let $\psi_i=(\alpha_i,\beta_i)$ and let $a_1(t,\psi_i)=0$ and  $a_2(t,\psi_i) = \alpha_i + \beta_i \, t$. Then, [[#categorical1|(1)]] gives a probability distribution for binary outcomes:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation= &amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij}=0 {{!}} \psi_i} = \displaystyle{\frac{1}{1 + e^{\alpha_i + \beta_i \, t_{ij} } } } \quad \ \ \ \text{and} \quad&lt;br /&gt;
\ \ \ \prob{y_{ij}=1 {{!}} \psi_i} = \displaystyle{\frac{e^{\alpha_i + \beta_i \, t_{ij} } }{1 + e^{\alpha_i + \beta_i \, t_{ij} } } }. &lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Such parametrizations are extremely flexible and easy to interpret in simple situations.&lt;br /&gt;
In the previous example for instance, $\prob{y_{ij}=1 | \psi_i}$ and $a_2(t_{ij},\psi_i)$ move in the same direction as time increases.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Ordinal data ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Ordinal data further assumes that the categories are ordered, i.e., there exists an order $\prec$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
c_1 \prec c_2,\prec \ldots \prec c_K .&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
We can think for instance of levels of pain (low, moderate, severe), or any scores on a discrete scale, e.g., from 1 to 10.&lt;br /&gt;
&lt;br /&gt;
Instead of defining the probabilities of each category, it may be convenient to define the cumulative probabilities $\prob{y_{ij} \preceq c_k | \psi_i}$ for $k=1,\ldots ,K-1$, or in the other direction: $\prob{y_{ij} \succeq c_k | \psi_i}$ for $k=2,\ldots, K$. &lt;br /&gt;
Any model is possible as long as it defines a probability distribution, i.e., satisfies:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
0 \leq  \prob{y_{ij} \preceq c_1 {{!}} \psi_i} \leq  \prob{y_{ij} \preceq c_2 {{!}} \bpsi_i} \leq \ldots \leq  \prob{y_{ij} \preceq c_K {{!}} \psi_i} =1 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Without any loss of generality, we will consider numerical categories in what follows. The order $\prec$  then reduces to the usual order $&amp;lt;$ on $\Rset$.&lt;br /&gt;
Currently, the most popular model for  ordinal data is the proportional odds model which uses ''logits'' of these cumulative probabilities, also called ''cumulative logits''. We assume that there exist $\alpha_{i,1}\geq0$, $\alpha_{i,2}\geq 0, \ldots , \alpha_{i,K-1}\geq 0$ such that for $k=1,2,\ldots,K-1$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq c_k {{!}} \psi_i} \right) = \left( \sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij}) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
where $x(t_{ij})$ is a vector of regression variables and $\beta_i$ a vector of coefficients. Here, $\bpsi_i=(\alpha_{i1},\alpha_{i2},\ldots,\alpha_{i,K-1},\beta_i)$.&lt;br /&gt;
&lt;br /&gt;
Recall that $\logit(p) = \log\left(p/(1-p)\right)$. Then, the probability defined in [[#propodds_model|(2)]] can also be expressed as&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij} \leq c_k {{!}} \bpsi_i}  = \displaystyle{\frac{1}{1 + e^{ \left(\sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij})} } }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= We give  to patients a drug which is supposed to decrease the level of a given type of pain. &lt;br /&gt;
The level of pain is measured on a scale from 1 to 3: 1=low, 2=moderate, 3=high. We consider the following model with the constraint that $\alpha_{i2}\geq 0$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 1 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \beta_{i,1}\, t_{ij} + \beta_{i,2}\, C_{ij} \\&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 2 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \alpha_{i,2} + \beta_{i,1}\, t_{ij} + \beta_{i2}\, C_{ij} \\&lt;br /&gt;
\prob{y_{ij} \leq 3 {{!}} \psi_i} &amp;amp;=&amp;amp; 1,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
where $C_{ij}$ is the concentration of the drug at time $t_{ij}$. The model parameters are quite easy to explain:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $\beta_{i,1}=0$   means that without treatment, the level of pain tends to remains stable  over time.&lt;br /&gt;
* $\beta_{i,1}&amp;lt;0$  (resp. $\beta_{i1}&amp;gt;0$) means that the pain tends to increase (resp. decrease) over time.&lt;br /&gt;
* $\beta_{i,2}=0$   means that the drug has no effect on pain.&lt;br /&gt;
* $\beta_{i,2}&amp;gt;0$   means that the level of pain tends to decrease when the  drug concentration increases, whereas $\beta_{i2}&amp;lt;0$ means that pain is an adverse drug effect.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Exclusive use of linear models (or generalized linear models) has no real justification today since very efficient tools are available for nonlinear models.&lt;br /&gt;
Model [[#propodds_model|(2)]] can be easily extended to a nonlinear model:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model2&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq k {{!}} \psi_i } \right) = \sum_{m=1}^k \alpha_{i,m} +  \beta(x(t_{ij})) , &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $\beta$ is any (linear or nonlinear) function of $x(t_{ij})$. }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Markovian dependence ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity, we will assume here that the observations $(y_{ij})$ take their values in $\{1, 2, \ldots, K\}$.&lt;br /&gt;
&lt;br /&gt;
We have so far assumed that the categorical observations $(y_{ij},\,j=1,2,\ldots,n_i)$ for individual $i$ are independent. It is however possible to introduce dependency between observations from the same individual by assuming that $(y_{ij},\,j=1,2,\ldots,n_i)$ forms a [http://en.wikipedia.org/wiki/Markov_chain Markov chain]. For instance, a Markov chain with memory 1 assumes that all is required from the past  to determine the distribution of $y_{i,j}$ is the value of  the previous observation $y_{i,j-1}$. i.e., for all $k=1,2,\ldots ,K$,&lt;br /&gt;
 &lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i,j} = k\, {{!}} \,y_{i,j-1}, y_{i,j-2}, y_{i,j-3},\ldots,\psi_i} = \prob{y_{i,j} = k {{!}} y_{i,j-1},\psi_i}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Discrete time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
If the observation times are regularly spaced (constant length of time between successive observations), we can consider the observations $(y_{ij},\,j=1,2,\ldots,n_i)$ to be a discrete time Markov chain. Here, for each individual $i$, the probability distribution of the sequence $(y_{ij},\,j=1,2,\ldots,n_i)$ is defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the distribution $  \pi_{i,1} = (\pi_{i,1}^{k} , k=1,2,\ldots,K)$ of the first observation $y_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \pi_{i,1}^{k} = \prob{y_{i,1} = k {{!}} \psi_i} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the sequence of ''transition matrices'' $(Q_{i,j}, j=2,3,\ldots)$, where for each $j$, $Q_{i,j} = (q_{i,j}^{\ell,k}, 1\leq \ell,k \leq K)$ is a matrix of size $K \times K$ such that,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
q_{i,j}^{\ell,k} &amp;amp;=&amp;amp; \prob{y_{i,j} = k {{!}} y_{i,j-1}=\ell , \psi_i} \quad \text{ for all } (\ell,k),\\&lt;br /&gt;
\sum_{k=1}^{K}q_{ij}^{\ell,k} &amp;amp;=&amp;amp; 1 \quad \text{ for all } (\ell,k).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution of  $y_i=(y_{i,j}, j=1,2,\ldots, n_i)$ is then well-defined:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \pmacro(y_{i,1}{{!}}\psi_i) \prod_{j=2}^{n_i} \pmacro(y_{i,j} {{!}} y_{i,j-1},\psi_i) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
For a given individual $i$,  $Q_{i,j}$ defines the transition probabilities between states at a given time $t_{ij}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:markov_1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Our model must therefore give, for each individual $i$, the distribution of first observation $(y_{i,1})$ and a description of how the transition probabilities evolve with time.&lt;br /&gt;
&lt;br /&gt;
The figure below shows several examples of simulated sequences coming from a model with 2 states defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit\left(q_{i,j}^{1,2}\right) &amp;amp;=&amp;amp; a_i+b_i \, t_j \\&lt;br /&gt;
\logit\left(q_{i,j}^{2,1}\right) &amp;amp;=&amp;amp; c_i+d_i \, t_j \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5 ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $t_j = j$.&lt;br /&gt;
&lt;br /&gt;
[[File:markov_2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
In the first example (left), the logits of the transitions between states are constant ($b_i = d_i = 0$).&lt;br /&gt;
Transition probabilities are therefore constant over time.  Here,  $q^{1,2}=1/(1+\exp(2.5))=0.0759$ and $q^{2,1}=1/(1+\exp(2))=0.1192$. As $q^{1,2}$ and $q^{2,1}$ are small with $q^{1,2}&amp;lt;q^{2,1}$, transitions between the two states are rare, and a larger amount of time (on average) is spent in  state 1. Indeed, the stationary distribution is the eigenvector of the transition matrix $P$: $\prob{y_{ij}=1}=0.611$ and $ \prob{y_{ij}=2}=0.389$.&lt;br /&gt;
The figure (left) displays the transition rates $q^{1,2}$ and $q^{2,1}$ as function of the time (top left) and two simulated sequences of states (centre and bottom left).&lt;br /&gt;
&lt;br /&gt;
In the second example (center), $b_i$ and $d_i$ are negative. This means that as time progresses, transitions from state 1 to 2 become rarer, and the same is true from 2 to 1.&lt;br /&gt;
&lt;br /&gt;
In the third example (right), now $b_i$ and $d_i$ are positive. This means that as time progresses, transitions from state 1 to 2 become more and more frequent, and also more frequent from 2 to 1.&lt;br /&gt;
Note that the value of $a_i$ (resp. $c_i$) can be seen as the transition probability from state 1 to 2 (resp. 2 to 1) at time $t=0$.&lt;br /&gt;
&lt;br /&gt;
Different choices can be made for defining an initial distribution $\pi_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The initial state can be defined arbitrarily: $y_{i,1}=k_0$. This means that $\pi_{i,1}^{k_0} = 1$ and $\pi_{i,1}^{k} = 0$ for $k\neq k_0$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* More generally, any simple probability distribution can be put on the choice of the initial state, e.g., the uniform distribution $\pi_{i,1}^{k} = 1/K$ for $ k=1,2,\ldots , K$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If a transition matrix $Q_{i1} $ has been defined at time $t_1$, we might consider using its stationary distribution, i.e., taking for $\pi_{i,1}$ the solution to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pi_{i,1} = \pi_{i,1} Q_{i1} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Continuous time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The previous situation can be extended to the case where observation times are irregular, by modeling the&lt;br /&gt;
sequence of states as a continuous-time [http://en.wikipedia.org/wiki/Markov_process Markov process]. The difference is that rather than transitioning to a new (possibly the same) state at each time step, the system remains in the current state for some random  amount of time before transitioning. This process is now  characterized by ''transition rates'' instead of transition probabilities:&lt;br /&gt;
&lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(t+h) = k\, {{!}} \,y_{i}(t)=\ell , \psi_i} = h \, \rho_{i}^{\ell,k}(t) + o(h),\quad k \neq \ell .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The probability that no transition happens between $t$ and $t+h$ is&lt;br /&gt;
 &lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(s) = \ell, \forall  s\in(t, t+h) \ {{!}}  \ y_{i}(t)=\ell , \psi_i} = e^{h \, \rho_{i}^{\ell,\ell}(t)} . &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------------------------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary&lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=  &lt;br /&gt;
A model for independent categorical data is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;The probability mass functions $\left(\prob{y_{ij} = k {{!}} \psi_i} \right)$&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative probability functions $\left(\prob{y_{ij} \leq c_k {{!}}  \psi_i} \right)$ for ordinal data&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative logits $\left(\logit \left( \prob{y_{ij} \leq k {{!}}  \psi_i} \right)\right)$ for a proportional odds model&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
A model for categorical data with Markovian dependency is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the probability transitions in the case of a discrete-time Markov chain&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the transition rates in the case of a continuous-time Markov process&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; the probability distribution of the initial states&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== $\mlxtran$ for categorical data models == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 1:&lt;br /&gt;
|title2= $ \quad y_{ij} \in \{0, 1, 2\}$&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (V_i, k_i, \alpha_{0,i}, \alpha_{1,i}, \gamma_i) \\[0.2cm]&lt;br /&gt;
D &amp;amp;=&amp;amp;100 \\&lt;br /&gt;
C(t,\psi_i) &amp;amp;=&amp;amp; \frac{D_i}{V_i} e^{-k_i \, t} \\[0.2cm]&lt;br /&gt;
\prob{y_{ij}\leq 0} &amp;amp;=&amp;amp; \alpha_{0,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 1} &amp;amp;=&amp;amp; \alpha_{0,i} + \alpha_{1,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 2} &amp;amp;=&amp;amp; 1&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;&lt;br /&gt;
INPUT:&lt;br /&gt;
input = {V, k, alpha0, alpha1, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
D = 100&lt;br /&gt;
C = D/V*exp(-k*t)&lt;br /&gt;
p0 = alpha0 + gamma*C&lt;br /&gt;
p1 = p0 + alpha1&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
y = {type=categorical,&lt;br /&gt;
     categories={0, 1, 2},&lt;br /&gt;
     P(y&amp;lt;=0)=p0,&lt;br /&gt;
     P(y&amp;lt;=1)=p1&lt;br /&gt;
     }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 2:&lt;br /&gt;
|title2= $\quad$ 2-state discrete-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i) \\[0.2cm]&lt;br /&gt;
\logit(p_{ij}^{12}) &amp;amp;=&amp;amp; a_i+b_i \, t_{ij} \\&lt;br /&gt;
\logit(p_{ij}^{21}) &amp;amp;=&amp;amp; c_i+d_i \, t_{ij} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = 0.5&lt;br /&gt;
      logit(P(Y=2 | Y_p=1)) = a + b*t&lt;br /&gt;
      logit(P(Y=1 | Y_p=2)) = c + d*t&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 3:&lt;br /&gt;
|title2= $\quad$ 2-state continuous-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i,\pi_i) \\[0.2cm]&lt;br /&gt;
q_{i}^{12}(t) &amp;amp;=&amp;amp; e^{a_i+b_i \, t} \\&lt;br /&gt;
q_{i}^{21}(t) &amp;amp;=&amp;amp; e^{c_i+d_i \, t} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; \pi_i&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d, pi}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = pi&lt;br /&gt;
      transitionRate(1,2) = exp(a + b*t)&lt;br /&gt;
      transitionRate(2,1) = exp(c + d*t)&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2010analysis,&lt;br /&gt;
  title={Analysis of ordinal categorical data},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={656},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2007introduction,&lt;br /&gt;
  title={An introduction to categorical data analysis},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={423},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bolker2009generalized,&lt;br /&gt;
  title={Generalized linear mixed models: a practical guide for ecology and evolution},&lt;br /&gt;
  author={Bolker, B. M. and Brooks, M. E. and Clark, C. J. and Geange, S. W. and Poulsen, J. R. and Stevens, M. H. H. and White, J.-S. S. and others},&lt;br /&gt;
  journal={Trends in ecology &amp;amp; evolution},&lt;br /&gt;
  volume={24},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={127-135},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Elsevier Science}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{davidian1995,&lt;br /&gt;
        author = {Davidian, M. and Giltinan, D. M.},&lt;br /&gt;
        title = {Nonlinear Models for Repeated Measurements Data },&lt;br /&gt;
        publisher = {Chapman &amp;amp; Hall.},&lt;br /&gt;
        address = {London},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        year = {1995}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{jiang2007,&lt;br /&gt;
        author = {Jiang., J.},&lt;br /&gt;
        title  = {Linear and Generalized Linear Mixed Models and Their Applications.},&lt;br /&gt;
        publisher = {Springer Series in Statistics},&lt;br /&gt;
        volume = {},&lt;br /&gt;
        pages = {},&lt;br /&gt;
        year = {2007},&lt;br /&gt;
        series = {},&lt;br /&gt;
        address = {New York},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        month = {}&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mcculloch2011generalized,&lt;br /&gt;
  title={Generalized, Linear, and Mixed Models},&lt;br /&gt;
  author={McCulloch, C. E. and Searle, S. R. and Neuhaus, J. M.},&lt;br /&gt;
  isbn={9781118209967},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
  url={http://books.google.fr/books?id=kyvgyK\_sBlkC},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{molenberghs2005models,&lt;br /&gt;
  title={Models for discrete longitudinal data},&lt;br /&gt;
  author={Molenberghs, G. and Verbeke, G.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{powers2008statistical,&lt;br /&gt;
  title={Statistical methods for categorical data analysis},&lt;br /&gt;
  author={Powers, D. A. and Xie, Y.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Emerald Group Publishing}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wolfinger1993generalized,&lt;br /&gt;
  title={Generalized linear mixed models a pseudo-likelihood approach},&lt;br /&gt;
  author={Wolfinger, R. and O'Connell, M.},&lt;br /&gt;
  journal={Journal of statistical Computation and Simulation},&lt;br /&gt;
  volume={48},&lt;br /&gt;
  number={3-4},&lt;br /&gt;
  pages={233-243},&lt;br /&gt;
  year={1993},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Models for count data&lt;br /&gt;
|linkNext=Models for time-to-event data }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7429</id>
		<title>Model for categorical data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Model_for_categorical_data&amp;diff=7429"/>
		<updated>2013-06-25T13:04:08Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Ordinal data */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
== Overview == &lt;br /&gt;
&lt;br /&gt;
Assume now that the observed data  takes its values in a fixed and finite set of nominal categories $\{c_1, c_2,\ldots , c_K\}$.&lt;br /&gt;
Considering the observations $(y_{ij}, 1 \leq j \leq n_i)$ of any individual $i$ as a sequence of independent random variables, the model is completely defined by the probability mass functions $\prob{y_{ij}=c_k | \psi_i}$, for $k=1,\ldots, K$ and $1 \leq j \leq n_i$.&lt;br /&gt;
&lt;br /&gt;
For a given $(i,j)$, the sum of the $K$ probabilities is 1, so in fact only $K-1$ of them need to be defined.&lt;br /&gt;
&lt;br /&gt;
In the most general way possible, any model can be considered so long as it defines a probability distribution, i.e., for each $k$,  $\prob{y_{ij}=c_k | \psi_i} \in [0,1]$, and $\sum_{k=1}^{K} \prob{y_{ij}=c_k | \psi_i} = 1$. For instance, we could define $K$ time-dependent parametric functions $a_1$, $a_2$, ..., $a_K$  and set for any individual $i$, time $t_{ij}$ and $k \in \{1,\ldots,K\}$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;categorical1&amp;quot; &amp;gt;&amp;lt;math&amp;gt; &lt;br /&gt;
\prob{y_{ij}=c_k {{!}} \psi_i} = \displaystyle{\frac{e^{a_k(t_{ij},\psi_i)} }{\sum_{m=1}^K e^{a_m(t_{ij},\psi_i)} } }.   &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(1) }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= Suppose we want to model binary data, i.e., data where $y_{ij} \in \{0,1\}$.&lt;br /&gt;
&lt;br /&gt;
Let $\psi_i=(\alpha_i,\beta_i)$ and let $a_1(t,\psi_i)=0$ and  $a_2(t,\psi_i) = \alpha_i + \beta_i \, t$. Then, [[#categorical1|(1)]] gives a probability distribution for binary outcomes:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation= &amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij}=0 {{!}} \psi_i} = \displaystyle{\frac{1}{1 + e^{\alpha_i + \beta_i \, t_{ij} } } } \quad \ \ \ \text{and} \quad&lt;br /&gt;
\ \ \ \prob{y_{ij}=1 {{!}} \psi_i} = \displaystyle{\frac{e^{\alpha_i + \beta_i \, t_{ij} } }{1 + e^{\alpha_i + \beta_i \, t_{ij} } } }. &lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Such parametrizations are extremely flexible and easy to interpret in simple situations.&lt;br /&gt;
In the previous example for instance, $\prob{y_{ij}=1 | \psi_i}$ and $a_2(t_{ij},\psi_i)$ move in the same direction as time increases.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== Ordinal data ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Ordinal data further assumes that the categories are ordered, i.e., there exists an order $\prec$ such that&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
c_1 \prec c_2,\prec \ldots \prec c_K .&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
We can think for instance of levels of pain (low, moderate, severe), or any scores on a discrete scale, e.g., from 1 to 10.&lt;br /&gt;
&lt;br /&gt;
Instead of defining the probabilities of each category, it may be convenient to define the cumulative probabilities $\prob{y_{ij} \preceq c_k | \psi_i}$ for $k=1,\ldots ,K-1$, or in the other direction: $\prob{y_{ij} \succeq c_k | \psi_i}$ for $k=2,\ldots, K$. &lt;br /&gt;
Any model is possible as long as it defines a probability distribution, i.e., satisfies:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
0 \leq  \prob{y_{ij} \preceq c_1 {{!}} \psi_i} \leq  \prob{y_{ij} \preceq c_2 {{!}} \bpsi_i} \leq \ldots \leq  \prob{y_{ij} \preceq c_K {{!}} \psi_i} =1 .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
Without any loss of generality, we will consider numerical categories in what follows. The order $\prec$  then reduces to the usual order $&amp;lt;$ on $\Rset$.&lt;br /&gt;
Currently, the most popular model for  ordinal data is the proportional odds model which uses ''logits'' of these cumulative probabilities, also called ''cumulative logits''. We assume that there exist $\alpha_{i,1}\geq0$, $\alpha_{i,2}\geq 0, \ldots , \alpha_{i,K-1}\geq 0$ such that for $k=1,2,\ldots,K-1$,&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq c_k {{!}} \psi_i} \right) = \left( \sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij}) ,&lt;br /&gt;
&amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(2) }}&lt;br /&gt;
&lt;br /&gt;
where $x(t_{ij})$ is a vector of regression variables and $\beta_i$ a vector of coefficients. Here, $\bpsi_i=(\alpha_{i1},\alpha_{i2},\ldots,\alpha_{i,K-1},\beta_i)$.&lt;br /&gt;
&lt;br /&gt;
Recall that $\logit(p) = \log\left(p/(1-p)\right)$. Then, the probability defined in [[#propodds_model|(2)]] can also be expressed as&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{ij} \leq c_k {{!}} \bpsi_i}  = \displaystyle{\frac{1}{1 + e^{ \left(\sum_{m=1}^k \alpha_{im}\right) +  \beta_i \, x(t_{ij})} } }.&lt;br /&gt;
&amp;lt;/math&amp;gt;}} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example:&lt;br /&gt;
|text= We give  to patients a drug which is supposed to decrease the level of a given type of pain. &lt;br /&gt;
The level of pain is measured on a scale from 1 to 3: 1=low, 2=moderate, 3=high. We consider the following model with the constraint that $\alpha_{i2}\geq 0$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 1 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \beta_{i,1}\, t_{ij} + \beta_{i,2}\, C_{ij} \\&lt;br /&gt;
\logit \left(\prob{y_{ij} \leq 2 {{!}} \psi_i}\right) &amp;amp;=&amp;amp;  \alpha_{i,1} + \alpha_{i,2} + \beta_{i,1}\, t_{ij} + \beta_{i2}\, C_{ij} \\&lt;br /&gt;
\prob{y_{ij} \leq 3 {{!}} \psi_i} &amp;amp;=&amp;amp; 1,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
where $C_{ij}$ is the concentration of the drug at time $t_{ij}$. The model parameters are quite easy to explain:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* $\beta_{i,1}=0$   means that without treatment, the level of pain tends to remains stable  over time.&lt;br /&gt;
* $\beta_{i,1}&amp;lt;0$  (resp. $\beta_{i1}&amp;gt;0$) means that the pain tends to increase (resp. decrease) over time.&lt;br /&gt;
* $\beta_{i,2}=0$   means that the drug has no effect on pain.&lt;br /&gt;
* $\beta_{i,2}&amp;gt;0$   means that the level of pain tends to decrease when the  drug concentration increases, whereas $\beta_{i2}&amp;lt;0$ means that pain is an adverse drug effect.&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Remarks&lt;br /&gt;
|title=Remarks&lt;br /&gt;
|text= Exclusive use of linear models (or generalized linear models) has no real justification today since very efficient tools are available for nonlinear models.&lt;br /&gt;
Model [[#propodds_model|(2)]] can be easily extended to a nonlinear model:&lt;br /&gt;
&lt;br /&gt;
{{EquationWithRef&lt;br /&gt;
|equation=&amp;lt;div id=&amp;quot;propodds_model2&amp;quot;&amp;gt;&amp;lt;math&amp;gt; \logit \left(\prob{y_{ij} \leq k {{!}} \psi_i } \right) = \sum_{m=1}^k \alpha_{i,m} +  \beta(x(t_{ij})) , &amp;lt;/math&amp;gt;&amp;lt;/div&amp;gt;&lt;br /&gt;
|reference=(3) }}&lt;br /&gt;
&lt;br /&gt;
where $\beta$ is any (linear or nonlinear) function of $x(t_{ij})$. }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Markovian dependence ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
For the sake of simplicity, we will assume here that the observations $(y_{ij})$ take their values in $\{1, 2, \ldots, K\}$.&lt;br /&gt;
&lt;br /&gt;
We have so far assumed that the categorical observations $(y_{ij},\,j=1,2,\ldots,n_i)$ for individual $i$ are independent. It is however possible to introduce dependency between observations from the same individual by assuming that $(y_{ij},\,j=1,2,\ldots,n_i)$ forms a [http://en.wikipedia.org/wiki/Markov_chain Markov chain]. For instance, a [http://en.wikipedia.org/wiki/Markov_chain Markov chain] with memory 1 assumes that all is required from the past  to determine the distribution of $y_{i,j}$ is the value of  the previous observation $y_{i,j-1}$. i.e., for all $k=1,2,\ldots ,K$,&lt;br /&gt;
 &lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i,j} = k\, {{!}} \,y_{i,j-1}, y_{i,j-2}, y_{i,j-3},\ldots,\psi_i} = \prob{y_{i,j} = k {{!}} y_{i,j-1},\psi_i}.&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Discrete time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
If the observation times are regularly spaced (constant length of time between successive observations), we can consider the observations $(y_{ij},\,j=1,2,\ldots,n_i)$ to be a discrete time [http://en.wikipedia.org/wiki/Markov_chain Markov chain]. Here, for each individual $i$, the probability distribution of the sequence $(y_{ij},\,j=1,2,\ldots,n_i)$ is defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the distribution $  \pi_{i,1} = (\pi_{i,1}^{k} , k=1,2,\ldots,K)$ of the first observation $y_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \pi_{i,1}^{k} = \prob{y_{i,1} = k {{!}} \psi_i} &amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* the sequence of ''transition matrices'' $(Q_{i,j}, j=2,3,\ldots)$, where for each $j$, $Q_{i,j} = (q_{i,j}^{\ell,k}, 1\leq \ell,k \leq K)$ is a matrix of size $K \times K$ such that,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
q_{i,j}^{\ell,k} &amp;amp;=&amp;amp; \prob{y_{i,j} = k {{!}} y_{i,j-1}=\ell , \psi_i} \quad \text{ for all } (\ell,k),\\&lt;br /&gt;
\sum_{k=1}^{K}q_{ij}^{\ell,k} &amp;amp;=&amp;amp; 1 \quad \text{ for all } (\ell,k).&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution of  $y_i=(y_{i,j}, j=1,2,\ldots, n_i)$ is then well-defined:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pcyipsii(y_i {{!}} \psi_i) = \pmacro(y_{i,1}{{!}}\psi_i) \prod_{j=2}^{n_i} \pmacro(y_{i,j} {{!}} y_{i,j-1},\psi_i) .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
For a given individual $i$,  $Q_{i,j}$ defines the transition probabilities between states at a given time $t_{ij}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:markov_1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Our model must therefore give, for each individual $i$, the distribution of first observation $(y_{i,1})$ and a description of how the transition probabilities evolve with time.&lt;br /&gt;
&lt;br /&gt;
The figure below shows several examples of simulated sequences coming from a model with 2 states defined by:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\begin{eqnarray}&lt;br /&gt;
\logit\left(q_{i,j}^{1,2}\right) &amp;amp;=&amp;amp; a_i+b_i \, t_j \\&lt;br /&gt;
\logit\left(q_{i,j}^{2,1}\right) &amp;amp;=&amp;amp; c_i+d_i \, t_j \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5 ,&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
where $t_j = j$.&lt;br /&gt;
&lt;br /&gt;
[[File:markov_2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
In the first example (left), the logits of the transitions between states are constant ($b_i = d_i = 0$).&lt;br /&gt;
Transition probabilities are therefore constant over time.  Here,  $q^{1,2}=1/(1+\exp(2.5))=0.0759$ and $q^{2,1}=1/(1+\exp(2))=0.1192$. As $q^{1,2}$ and $q^{2,1}$ are small with $q^{1,2}&amp;lt;q^{2,1}$, transitions between the two states are rare, and a larger amount of time (on average) is spent in  state 1. Indeed, the stationary distribution is the eigenvector of the transition matrix $P$: $\prob{y_{ij}=1}=0.611$ and $ \prob{y_{ij}=2}=0.389$.&lt;br /&gt;
The figure (left) displays the transition rates $q^{1,2}$ and $q^{2,1}$ as function of the time (top left) and two simulated sequences of states (centre and bottom left).&lt;br /&gt;
&lt;br /&gt;
In the second example (center), $b_i$ and $d_i$ are negative. This means that as time progresses, transitions from state 1 to 2 become rarer, and the same is true from 2 to 1.&lt;br /&gt;
&lt;br /&gt;
In the third example (right), now $b_i$ and $d_i$ are positive. This means that as time progresses, transitions from state 1 to 2 become more and more frequent, and also more frequent from 2 to 1.&lt;br /&gt;
Note that the value of $a_i$ (resp. $c_i$) can be seen as the transition probability from state 1 to 2 (resp. 2 to 1) at time $t=0$.&lt;br /&gt;
&lt;br /&gt;
Different choices can be made for defining an initial distribution $\pi_{i,1}$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* The initial state can be defined arbitrarily: $y_{i,1}=k_0$. This means that $\pi_{i,1}^{k_0} = 1$ and $\pi_{i,1}^{k} = 0$ for $k\neq k_0$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* More generally, any simple probability distribution can be put on the choice of the initial state, e.g., the uniform distribution $\pi_{i,1}^{k} = 1/K$ for $ k=1,2,\ldots , K$.&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If a transition matrix $Q_{i1} $ has been defined at time $t_1$, we might consider using its stationary distribution, i.e., taking for $\pi_{i,1}$ the solution to:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\pi_{i,1} = \pi_{i,1} Q_{i1} .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Continuous time Markov chains ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The previous situation can be extended to the case where observation times are irregular, by modeling the&lt;br /&gt;
sequence of states as a continuous-time [http://en.wikipedia.org/wiki/Markov_process Markov process]. The difference is that rather than transitioning to a new (possibly the same) state at each time step, the system remains in the current state for some random  amount of time before transitioning. This process is now  characterized by ''transition rates'' instead of transition probabilities:&lt;br /&gt;
&lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(t+h) = k\, {{!}} \,y_{i}(t)=\ell , \psi_i} = h \, \rho_{i}^{\ell,k}(t) + o(h),\quad k \neq \ell .&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
The probability that no transition happens between $t$ and $t+h$ is&lt;br /&gt;
 &lt;br /&gt;
{{Equation1 &lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y_{i}(s) = \ell, \forall  s\in(t, t+h) \ {{!}}  \ y_{i}(t)=\ell , \psi_i} = e^{h \, \rho_{i}^{\ell,\ell}(t)} . &lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
------------------------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary&lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=  &lt;br /&gt;
A model for independent categorical data is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt;The probability mass functions $\left(\prob{y_{ij} = k {{!}} \psi_i} \right)$&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative probability functions $\left(\prob{y_{ij} \leq c_k {{!}}  \psi_i} \right)$ for ordinal data&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the cumulative logits $\left(\logit \left( \prob{y_{ij} \leq k {{!}}  \psi_i} \right)\right)$ for a proportional odds model&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
A model for categorical data with Markovian dependency is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ol&amp;gt;&lt;br /&gt;
&amp;lt;li&amp;gt; the probability transitions in the case of a discrete-time [http://en.wikipedia.org/wiki/Markov_chain Markov chain]&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; (or) the transition rates in the case of a continuous-time [http://en.wikipedia.org/wiki/Markov_process Markov process]&amp;lt;/li&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;li&amp;gt; the probability distribution of the initial states&amp;lt;/li&amp;gt;&lt;br /&gt;
&amp;lt;/ol&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== $\mlxtran$ for categorical data models == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 1:&lt;br /&gt;
|title2= $ \quad y_{ij} \in \{0, 1, 2\}$&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (V_i, k_i, \alpha_{0,i}, \alpha_{1,i}, \gamma_i) \\[0.2cm]&lt;br /&gt;
D &amp;amp;=&amp;amp;100 \\&lt;br /&gt;
C(t,\psi_i) &amp;amp;=&amp;amp; \frac{D_i}{V_i} e^{-k_i \, t} \\[0.2cm]&lt;br /&gt;
\prob{y_{ij}\leq 0} &amp;amp;=&amp;amp; \alpha_{0,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 1} &amp;amp;=&amp;amp; \alpha_{0,i} + \alpha_{1,i} + \gamma_i \, C(t_{ij},\psi_i) \\&lt;br /&gt;
\prob{y_{ij}\leq 2} &amp;amp;=&amp;amp; 1&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt;&lt;br /&gt;
INPUT:&lt;br /&gt;
input = {V, k, alpha0, alpha1, gamma}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
D = 100&lt;br /&gt;
C = D/V*exp(-k*t)&lt;br /&gt;
p0 = alpha0 + gamma*C&lt;br /&gt;
p1 = p0 + alpha1&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
y = {type=categorical,&lt;br /&gt;
     categories={0, 1, 2},&lt;br /&gt;
     P(y&amp;lt;=0)=p0,&lt;br /&gt;
     P(y&amp;lt;=1)=p1&lt;br /&gt;
     }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 2:&lt;br /&gt;
|title2= $\quad$ 2-state discrete-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i) \\[0.2cm]&lt;br /&gt;
\logit(p_{ij}^{12}) &amp;amp;=&amp;amp; a_i+b_i \, t_{ij} \\&lt;br /&gt;
\logit(p_{ij}^{21}) &amp;amp;=&amp;amp; c_i+d_i \, t_{ij} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; 0.5&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = 0.5&lt;br /&gt;
      logit(P(Y=2 | Y_p=1)) = a + b*t&lt;br /&gt;
      logit(P(Y=1 | Y_p=2)) = c + d*t&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1=Example 3:&lt;br /&gt;
|title2= $\quad$ 2-state continuous-time Markov chain&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (a_i,b_i,c_i,d_i,\pi_i) \\[0.2cm]&lt;br /&gt;
q_{i}^{12}(t) &amp;amp;=&amp;amp; e^{a_i+b_i \, t} \\&lt;br /&gt;
q_{i}^{21}(t) &amp;amp;=&amp;amp; e^{c_i+d_i \, t} \\&lt;br /&gt;
\prob{y_{i,1}=1} &amp;amp;=&amp;amp; \pi_i&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {a, b, c, d, pi}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = { type = categorical,&lt;br /&gt;
      categories = {1, 2},&lt;br /&gt;
      dependence = Markov&lt;br /&gt;
      P(Y_1=1) = pi&lt;br /&gt;
      transitionRate(1,2) = exp(a + b*t)&lt;br /&gt;
      transitionRate(2,1) = exp(c + d*t)&lt;br /&gt;
      }&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2010analysis,&lt;br /&gt;
  title={Analysis of ordinal categorical data},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={656},&lt;br /&gt;
  year={2010},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{agresti2007introduction,&lt;br /&gt;
  title={An introduction to categorical data analysis},&lt;br /&gt;
  author={Agresti, A.},&lt;br /&gt;
  volume={423},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Wiley-Interscience}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bolker2009generalized,&lt;br /&gt;
  title={Generalized linear mixed models: a practical guide for ecology and evolution},&lt;br /&gt;
  author={Bolker, B. M. and Brooks, M. E. and Clark, C. J. and Geange, S. W. and Poulsen, J. R. and Stevens, M. H. H. and White, J.-S. S. and others},&lt;br /&gt;
  journal={Trends in ecology &amp;amp; evolution},&lt;br /&gt;
  volume={24},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={127-135},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Elsevier Science}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{davidian1995,&lt;br /&gt;
        author = {Davidian, M. and Giltinan, D. M.},&lt;br /&gt;
        title = {Nonlinear Models for Repeated Measurements Data },&lt;br /&gt;
        publisher = {Chapman &amp;amp; Hall.},&lt;br /&gt;
        address = {London},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        year = {1995}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{jiang2007,&lt;br /&gt;
        author = {Jiang., J.},&lt;br /&gt;
        title  = {Linear and Generalized Linear Mixed Models and Their Applications.},&lt;br /&gt;
        publisher = {Springer Series in Statistics},&lt;br /&gt;
        volume = {},&lt;br /&gt;
        pages = {},&lt;br /&gt;
        year = {2007},&lt;br /&gt;
        series = {},&lt;br /&gt;
        address = {New York},&lt;br /&gt;
        edition = {},&lt;br /&gt;
        month = {}&lt;br /&gt;
}&lt;br /&gt;
&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{littell2006sas,&lt;br /&gt;
  title={SAS for mixed models},&lt;br /&gt;
  author={Littell, R. C.},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAS institute}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mcculloch2011generalized,&lt;br /&gt;
  title={Generalized, Linear, and Mixed Models},&lt;br /&gt;
  author={McCulloch, C. E. and Searle, S. R. and Neuhaus, J. M.},&lt;br /&gt;
  isbn={9781118209967},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
  url={http://books.google.fr/books?id=kyvgyK\_sBlkC},&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{molenberghs2005models,&lt;br /&gt;
  title={Models for discrete longitudinal data},&lt;br /&gt;
  author={Molenberghs, G. and Verbeke, G.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{powers2008statistical,&lt;br /&gt;
  title={Statistical methods for categorical data analysis},&lt;br /&gt;
  author={Powers, D. A. and Xie, Y.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Emerald Group Publishing}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wolfinger1993generalized,&lt;br /&gt;
  title={Generalized linear mixed models a pseudo-likelihood approach},&lt;br /&gt;
  author={Wolfinger, R. and O'Connell, M.},&lt;br /&gt;
  journal={Journal of statistical Computation and Simulation},&lt;br /&gt;
  volume={48},&lt;br /&gt;
  number={3-4},&lt;br /&gt;
  pages={233-243},&lt;br /&gt;
  year={1993},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Models for count data&lt;br /&gt;
|linkNext=Models for time-to-event data }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_count_data&amp;diff=7428</id>
		<title>Models for count data</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Models_for_count_data&amp;diff=7428"/>
		<updated>2013-06-25T13:00:37Z</updated>

		<summary type="html">&lt;p&gt;Admin: &lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;&amp;lt;!-- Menu for the Observations chapter --&amp;gt;&lt;br /&gt;
&amp;lt;sidebarmenu&amp;gt;&lt;br /&gt;
+[[Modeling the observations]]&lt;br /&gt;
*[[Modeling the observations| Introduction ]] | [[ Continuous data models ]] | [[Models for count data]]  | [[Model for categorical data]]  | [[Models for time-to-event data ]] | [[Joint models]]  &lt;br /&gt;
&amp;lt;/sidebarmenu&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Count data is a special type of statistical data that can only take non-negative integer values $\{0, 1, 2,\ldots\}$ that come from counting something, e.g., the number of [http://en.wikipedia.org/wiki/Seizures seizures], [http://en.wikipedia.org/wiki/Hemorrhages hemorrhages] or lesions in each given time period. More precisely, data from individual $i$ is the sequence $y_i=(y_{ij},1\leq j \leq n_i)$ where $y_{ij}$ is the number of events observed in the $j$th time interval $I_{ij}$.&lt;br /&gt;
&lt;br /&gt;
For the moment, let us assume that all the intervals have the same length. This is the case, for instance, if data are daily seizure counts: $I_{ij}$ is the $j$th day after the start of the experiment and $y_{ij}$ the number of seizures observed during that day.&lt;br /&gt;
&lt;br /&gt;
We will then model the sequence  $y_i=(y_{ij},1\leq j \leq n_i)$ as a sequence of  random variables that  take its values in $\{ 0, 1, 2,\ldots\}$.&lt;br /&gt;
&lt;br /&gt;
If we assume that these random variables are independent, then the model is completely defined by the [http://en.wikipedia.org/wiki/Probability_mass_function probability mass functions] $\prob{y_{ij}=k}$, for $k \geq 0$ and $1 \leq j \leq n_i$. Common distributions used to model count data include [http://en.wikipedia.org/wiki/Poisson_distribution Poisson], [http://en.wikipedia.org/wiki/Binomial_distribution binomial] and [http://en.wikipedia.org/wiki/Negative_binomial_distribution negative binomial].&lt;br /&gt;
&lt;br /&gt;
Indeed, here we will only consider [http://en.wikipedia.org/wiki/Parametric_model parametric distributions]. In this context, building a model means defining:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* the parameter function (or &amp;quot;intensity&amp;quot;) $\lambda_{ij} = \lambda(t_{ij},\psi_i)$ for any individual $i$ that depends on individual parameters $\psi_i$ and possibly the time $t_{ij}$.&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* the probability mass function $\prob{y_{ij}=k; \lambda_{ij}}$.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
The conditional distribution of the observations is therefore written:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation = &amp;lt;math&amp;gt; \prob{y_{ij}=k {{!}} \psi_i} = \prob{y_{ij}=k ; \lambda_{ij} }. &amp;lt;/math&amp;gt; }} &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Example&lt;br /&gt;
|title=Example&lt;br /&gt;
&lt;br /&gt;
|text= Let us illustrate this approach for the Poisson distribution.&lt;br /&gt;
A Poisson distribution with intensity $\lambda$ is defined by its probability mass function:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt; \prob{y=k  ; \lambda} = \displaystyle{\frac{\lambda^{k} \, e^{-\lambda} }{k!} }. &amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:poisson1.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
One of the main property of the Poisson distribution is that $\lambda$ is both the mean and the variance of the distribution:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt;\esp{y} = \var{y} = \lambda &amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
All that remains is to define the Poisson intensity function $ \lambda_{ij} = \lambda(t_{ij},\psi_i)$. Then,&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;\prob{y_{ij}=k {{!}} \psi_i} = \displaystyle{\frac{\lambda_{ij}^{k}\, e^{-\lambda_{ij} } } {k!} }. &amp;lt;/math&amp;gt;}}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
There are many variations of  the Poisson model:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* ''Homogeneous Poisson distribution:'' this assumes a constant intensity $\lambda_i$ for each individual $i$. Here,  $\psi_i = \lambda_i$ and $\lambda(t_{ij},\psi_i)=\lambda_i$. &lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* ''Non-homogeneous Poisson distribution:'' this assumes that the Poisson intensity is a function of time. For example, suppose that we believe that a disease-related event is increasing linearly in frequency each month. We could then model this using $\lambda(t_{ij},\psi_i) =  \lambda_{i} + a_i t_{ij}$, where $t_{ij} = j$ (months). Here, $\psi_i=(\lambda_{i},a_i)$.&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
* ''Additional regression variables:'' the Poisson intensity may depend on regression variables other than time. For example, assume  that taking a drug tends to reduce the number of events. We can then link the time-varying drug concentration $C$ to the value of $\lambda$ at time $t_{ij}$ using for instance an &amp;quot;Imax&amp;quot; model:&lt;br /&gt;
&lt;br /&gt;
{{Equation1|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\lambda(t_{ij},\psi_i) = \lambda_{i}\left(1-\Imax_i\displaystyle{\frac{ \ C_i(t_{ij})}{IC_{50,i} +  C_i(t_{ij})} }\right) ,&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
: where $\lambda_{i}$ is the baseline intensity and where $0\leq \Imax_i\leq 1$. Here, $\psi_{i} = (\lambda_{i}, \Imax_i, IC_{50,i})$.&lt;br /&gt;
&lt;br /&gt;
: This model can even be combined with the previous non-homogeneous model by assuming a time-varying baseline  $\lambda_{i}(t)$ in order to combine a drug effect model with a disease model for instance.&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* Instead of assuming independent count data, we can introduce Markovian dependency into the model by assuming for example that $\lambda_{ij}$ is function of $y_{i,j-1}$. Then, $\prob{y_{ij}=k\, |\, y_{i\,j-1}, t_{ij},\psi_i}$ is the probability function of a Poisson random variable with parameter $\lambda_{ij} =\lambda(y_{i,j-1}, t_{ij},\psi_i)$.&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
* If $y_{ij}$ is the number of a given type of events (seizures, hemorrhages, etc.) in a given time interval $I_{ij}$, and if $h_i(t)=h(t,\psi_i)$ is the hazard function associated with this sequence of events for individual $i$, then $y_{ij}$ is a non-homogeneous Poisson process with  Poisson intensity $\lambda_{ij}=\displaystyle{ \int_{I_{ij}}} h(t,\psi_i)dt$ in interval $I_{ij}$ (see [[Models for time-to-event data]] section).&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
Let us see now some other examples of distributions for count data:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
* Poisson distribution:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; &lt;br /&gt;
\prob{y=k  ; \lambda,p_0} = \left\{ \begin{array}{cc}&lt;br /&gt;
p_0 + (1-p_0)e^{-\lambda} &amp;amp; {\rm if } \  k=0 \\&lt;br /&gt;
(1-p_0) \displaystyle {\frac{e^{-\lambda} \lambda^{k} }{k!} } &amp;amp; {\rm if } \  k&amp;gt;0 .&lt;br /&gt;
 \end{array}&lt;br /&gt;
\right.&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
:where $0\leq p_0 &amp;lt;1$. This is useful when data seem generally to follow a Poisson distribution except for having an overly large quantity of cases when $k=0$:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:poisson2.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* The negative binomial distribution is:&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y=k  ; p,r} = \displaystyle{ \frac{\Gamma(k+r)}{k!\, \Gamma(r)} }(1-p)^r p^k ,&lt;br /&gt;
&amp;lt;/math&amp;gt;}}&lt;br /&gt;
&lt;br /&gt;
:with $0\leq p \leq 1$ and $r&amp;gt;0$. If $r$ is an integer, then the negative binomial (NB) distribution with parameters $(p,r)$ is the probability distribution of the number of successes in a sequence of [http://en.wikipedia.org/wiki/Bernoulli_trial Bernoulli trials] with probability of success $p$ before $r$ failures occur.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:poisson3.png|link=]]&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
* The generalized Poisson distribution is: &lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt;&lt;br /&gt;
\prob{y=k  ; \lambda,\delta} = \displaystyle {\frac{\lambda (\lambda+k\delta)^{k-1} e^{-\lambda-k\delta} }{k!} },&lt;br /&gt;
&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
:with $\lambda&amp;gt;0$ and $0\leq \delta &amp;lt;1$.&lt;br /&gt;
:The generalized Poisson (GP) distribution includes the Poisson distribution as a special case $(\delta=0)$, and is over-dispersed relative to the Poisson. Indeed, the variance to mean ratio exceeds 1:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{Equation1&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{eqnarray} \esp{y} &amp;amp;=&amp;amp; \frac{\lambda}{1-\delta} \\&lt;br /&gt;
\var{y} &amp;amp;=&amp;amp; \frac{\lambda}{1-\delta^3}.&lt;br /&gt;
\end{eqnarray}&amp;lt;/math&amp;gt; }}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
::[[File:poisson4.png|link=]]&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
-----------------&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Summary&lt;br /&gt;
|title=Summary&lt;br /&gt;
|text=&lt;br /&gt;
For a given design $\bx_{i}$ and a given vector of parameters $\psi_i$, a parametric model for count data is completely defined by:&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;ul&amp;gt;&lt;br /&gt;
- the probability mass function used to represent the distribution of the data in a given time interval&lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
- a model which defines how the distribution's parameter function (i.e., intensity) varies over time.&lt;br /&gt;
&amp;lt;/ul&amp;gt;&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
== $\mlxtran$ for count data models == &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1= Example 1: &lt;br /&gt;
|title2= Poisson model with time varying intensity&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{array}{c}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (\alpha_i,\beta_i) \\[0.3cm]&lt;br /&gt;
\lambda(t,\psi_i) &amp;amp;=&amp;amp; \alpha_i + \beta_i\,t \\[0.3cm]&lt;br /&gt;
\prob{y_{ij}=k} &amp;amp;=&amp;amp; \displaystyle{ \frac{\lambda(t_{ij} , \psi_i)^k}{k!} } e^{-\lambda(t_{ij} , \psi_i)}\\&lt;br /&gt;
\end{array}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF; border: none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
input = {alpha, beta}&lt;br /&gt;
&lt;br /&gt;
EQUATION:&lt;br /&gt;
lambda = alpha + beta*t&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
y ~ poisson(lambda)&lt;br /&gt;
&amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
{{ExampleWithCode&lt;br /&gt;
|title1= Example 2:  &lt;br /&gt;
|title2= generalized Poisson model&lt;br /&gt;
|text=&lt;br /&gt;
&lt;br /&gt;
|equation=&amp;lt;math&amp;gt; \begin{array}{c}&lt;br /&gt;
\psi_i &amp;amp;=&amp;amp; (\lambda_i,\delta_i) \\&lt;br /&gt;
\log\left( \prob{y_{ij}=k} \right) &amp;amp;=&amp;amp; \log(\lambda_i) +  (k-1)\log(\lambda_i+k\delta_i) \\&lt;br /&gt;
&amp;amp;&amp;amp; -\lambda_i-k\delta_i - \log(k!)\\[1cm]&lt;br /&gt;
\end{array}&amp;lt;/math&amp;gt;&lt;br /&gt;
|code=&lt;br /&gt;
{{MLXTranForTable&lt;br /&gt;
|name=&lt;br /&gt;
|text=&lt;br /&gt;
&amp;lt;pre style=&amp;quot; background-color:#EFEFEF;  border:none;&amp;quot;&amp;gt; &lt;br /&gt;
INPUT:&lt;br /&gt;
parameter = {dlt, lbd}&lt;br /&gt;
&lt;br /&gt;
DEFINITION:&lt;br /&gt;
Y = {&lt;br /&gt;
  type = count,&lt;br /&gt;
  log(P(Y=k)) = log(lambda)&lt;br /&gt;
  + (k-1)*log(lambda+k*delta)&lt;br /&gt;
  - lambda -k*delta - factln(k)&lt;br /&gt;
} &amp;lt;/pre&amp;gt; }}&lt;br /&gt;
}}&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Bibliography==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{blundell2002individual,&lt;br /&gt;
  title={Individual effects and dynamics in count data models},&lt;br /&gt;
  author={Blundell, R. and Griffith, R. and Windmeijer, F.},&lt;br /&gt;
  journal={Journal of Econometrics},&lt;br /&gt;
  volume={108},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={113-131},&lt;br /&gt;
  year={2002},&lt;br /&gt;
  publisher={Elsevier}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{bolker2009generalized,&lt;br /&gt;
  title={Generalized linear mixed models: a practical guide for ecology and evolution},&lt;br /&gt;
  author={Bolker, B. M. and Brooks, M. E. and Clark, C. J. and Geange, S. W. and Poulsen, J. R. and Stevens, M. H. and White, J.-S. S. and others},&lt;br /&gt;
  journal={Trends in ecology &amp;amp; evolution},&lt;br /&gt;
  volume={24},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={127-135},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Elsevier Science}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{cameron1998regression,&lt;br /&gt;
  title={Regression analysis of count data},&lt;br /&gt;
  author={Cameron, A. C. and Trivedi, P. K.},&lt;br /&gt;
  volume={30},&lt;br /&gt;
  year={1998},&lt;br /&gt;
  publisher={Cambridge University Press}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{christensen2002bayesian,&lt;br /&gt;
  title={Bayesian prediction of spatial count data using generalized linear mixed models},&lt;br /&gt;
  author={Christensen, O. F. and Waagepetersen, R.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={58},&lt;br /&gt;
  number={2},&lt;br /&gt;
  pages={280-286},&lt;br /&gt;
  year={2002},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{fahrmeir1994multivariate,&lt;br /&gt;
  title={Multivariate statistical modelling based on generalized linear models},&lt;br /&gt;
  author={Fahrmeir, L. and Tutz, G. and Hennevogl, W.},&lt;br /&gt;
  volume={2},&lt;br /&gt;
  year={1994},&lt;br /&gt;
  publisher={Springer New York}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{hall2004zero,&lt;br /&gt;
  title={Zero-inflated Poisson and binomial regression with random effects: a case study},&lt;br /&gt;
  author={Hall, D. B.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  volume={56},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={103--1039},&lt;br /&gt;
  year={2004},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{heilbron2007zero,&lt;br /&gt;
  title={Zero-Altered and other Regression Models for Count Data with Added Zeros},&lt;br /&gt;
  author={Heilbron, D. C.},&lt;br /&gt;
  journal={Biometrical Journal},&lt;br /&gt;
  volume={36},&lt;br /&gt;
  number={5},&lt;br /&gt;
  pages={531-547},&lt;br /&gt;
  year={2007},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{lawless1987negative,&lt;br /&gt;
  title={Negative binomial and mixed Poisson regression},&lt;br /&gt;
  author={Lawless, J. F.},&lt;br /&gt;
  journal={Canadian Journal of Statistics},&lt;br /&gt;
  volume={15},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={209-225},&lt;br /&gt;
  year={1987},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{lee2006multi,&lt;br /&gt;
  title={Multi-level zero-inflated Poisson regression modelling of correlated count data with excess zeros},&lt;br /&gt;
  author={Lee, A. H. and Wang, K. and Scott, J. A. and Yau, K. K. W. and McLachlan, G. J.},&lt;br /&gt;
  journal={Statistical Methods in Medical Research},&lt;br /&gt;
  volume={15},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={47-61},&lt;br /&gt;
  year={2006},&lt;br /&gt;
  publisher={SAGE Publications}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{mcculloch2011generalized,&lt;br /&gt;
  title={Generalized, Linear, and Mixed Models},&lt;br /&gt;
  author={McCulloch, C. E. and Searle, S. R. and Neuhaus, J. M.},&lt;br /&gt;
  isbn={9781118209967},&lt;br /&gt;
  series={Wiley Series in Probability and Statistics},&lt;br /&gt;
  url={http://books.google.fr/books?id=kyvgyK\_sBlkC},&lt;br /&gt;
  year={2011},&lt;br /&gt;
  publisher={Wiley}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{min2005random,&lt;br /&gt;
  title={Random effect models for repeated measures of zero-inflated count data},&lt;br /&gt;
  author={Min, Y. and Agresti, A.},&lt;br /&gt;
  journal={Statistical Modelling},&lt;br /&gt;
  volume={5},&lt;br /&gt;
  number={1},&lt;br /&gt;
  pages={1-19},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={SAGE Publications}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{molenberghs2005models,&lt;br /&gt;
  title={Models for discrete longitudinal data},&lt;br /&gt;
  author={Molenberghs, G. and Verbeke, G.},&lt;br /&gt;
  year={2005},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{mullahy1998heterogeneity,&lt;br /&gt;
  title={Heterogeneity, excess zeros, and the structure of count data models},&lt;br /&gt;
  author={Mullahy, J.},&lt;br /&gt;
  journal={Journal of Applied Econometrics},&lt;br /&gt;
  volume={12},&lt;br /&gt;
  number={3},&lt;br /&gt;
  pages={337-350},&lt;br /&gt;
  year={1998},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{savic2009performance,&lt;br /&gt;
  title={Performance in population models for count data, part ii: A new saem algorithm},&lt;br /&gt;
  author={Savic, R. and Lavielle, M.},&lt;br /&gt;
  journal={Journal of pharmacokinetics and pharmacodynamics},&lt;br /&gt;
  volume={36},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={367-379},&lt;br /&gt;
  year={2009},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{thall1988mixed,&lt;br /&gt;
  title={Mixed Poisson likelihood regression models for longitudinal interval count data},&lt;br /&gt;
  author={Thall, P. F.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  pages={197-209},&lt;br /&gt;
  year={1988},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{thall1990some,&lt;br /&gt;
  title={Some covariance models for longitudinal count data with overdispersion},&lt;br /&gt;
  author={Thall, P. F. and Vail, S. C.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  pages={657-671},&lt;br /&gt;
  year={1990},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{tempelman1996mixed,&lt;br /&gt;
  title={A mixed effects model for overdispersed count data in animal breeding},&lt;br /&gt;
  author={Tempelman, R. J. and Gianola, D.},&lt;br /&gt;
  journal={Biometrics},&lt;br /&gt;
  pages={265-279},&lt;br /&gt;
  year={1996},&lt;br /&gt;
  publisher={JSTOR}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@book{winkelmann2008econometric,&lt;br /&gt;
  title={Econometric analysis of count data},&lt;br /&gt;
  author={Winkelmann, R.},&lt;br /&gt;
  year={2008},&lt;br /&gt;
  publisher={Springer}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{wolfinger1993generalized,&lt;br /&gt;
  title={Generalized linear mixed models a pseudo-likelihood approach},&lt;br /&gt;
  author={Wolfinger, R. and O'Connell, M.},&lt;br /&gt;
  journal={Journal of statistical Computation and Simulation},&lt;br /&gt;
  volume={48},&lt;br /&gt;
  number={3-4},&lt;br /&gt;
  pages={233-243},&lt;br /&gt;
  year={1993},&lt;br /&gt;
  publisher={Taylor &amp;amp; Francis}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{yau2003zero,&lt;br /&gt;
  title={Zero-Inflated Negative Binomial Mixed Regression Modeling of Over-Dispersed Count Data with Extra Zeros},&lt;br /&gt;
  author={Yau, K. K. W. and Wang, K. and Lee, A. H.},&lt;br /&gt;
  journal={Biometrical Journal},&lt;br /&gt;
  volume={45},&lt;br /&gt;
  number={4},&lt;br /&gt;
  pages={437-452},&lt;br /&gt;
  year={2003},&lt;br /&gt;
  publisher={Wiley Online Library}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&amp;lt;bibtex&amp;gt;&lt;br /&gt;
@article{zeileis2008regression,&lt;br /&gt;
  title={Regression models for count data in R},&lt;br /&gt;
  author={Zeileis, A. and Kleiber, C. and Jackman, S.},&lt;br /&gt;
  journal={Journal of Statistical Software},&lt;br /&gt;
  volume={27},&lt;br /&gt;
  number={8},&lt;br /&gt;
  pages={1-25},&lt;br /&gt;
  year={2008}&lt;br /&gt;
}&lt;br /&gt;
&amp;lt;/bibtex&amp;gt;&lt;br /&gt;
&lt;br /&gt;
{{Back&amp;amp;Next&lt;br /&gt;
|linkBack=Continuous data models&lt;br /&gt;
|linkNext=Model for categorical data }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
	<entry>
		<id>https://wiki.inria.fr/wikis/popix/index.php?title=Introduction_to_PK_modeling_using_MLXPlore_-_Part_I&amp;diff=7427</id>
		<title>Introduction to PK modeling using MLXPlore - Part I</title>
		<link rel="alternate" type="text/html" href="https://wiki.inria.fr/wikis/popix/index.php?title=Introduction_to_PK_modeling_using_MLXPlore_-_Part_I&amp;diff=7427"/>
		<updated>2013-06-25T12:27:53Z</updated>

		<summary type="html">&lt;p&gt;Admin: /* Introduction */&lt;/p&gt;
&lt;hr /&gt;
&lt;div&gt;== Introduction ==&lt;br /&gt;
This is an introductory tutorial for describing and visualizing simple and more complex [http://en.wikipedia.org/wiki/Pharmacokinetics pharmacokinetic] (PK) models.&lt;br /&gt;
&lt;br /&gt;
We will present several PK model examples and visualize the processes of [http://en.wikipedia.org/wiki/Absorption_%28pharmacokinetics%29 absorption], [http://en.wikipedia.org/wiki/Distribution_%28pharmacology%29 distribution] and [http://en.wikipedia.org/wiki/Elimination_%28pharmacology%29 elimination] that characterize them.  &lt;br /&gt;
&lt;br /&gt;
We will suppose in all these examples that a single dose is administered at time t=0.&lt;br /&gt;
In each example, the modeling goal is defined. Then, the model and requests for graphical outputs are coded in [http://www.lixoft.eu/products/mlxplore/mlxplore-overview/ MLXPlore], a new graphical and interactive software for the exploration and visualization of complex [http://en.wikipedia.org/wiki/Pharmacometrics pharmacometric] models. MLXPlore uses the easy and intuitive [http://www.lixoft.com/wp-content/resources/docs/modelMLXTRANtutorial.pdf MLXtran] model coding language, popularized by the [http://ww35.monolix.org/ Monolix] software. &lt;br /&gt;
&lt;br /&gt;
MLXPlore is used here for computing the predicted amount in the central compartment. We further display in the [[ Introduction to PK modeling using MLXPlore - Part II | Part II]] the predicted amount in the depot compartment and the MLXPlore project that was used for computing it.&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Absorption ==&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
=== First-order and zero-order absorption ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|absorption1a_script:}}&amp;lt;/span&amp;gt; this computes and displays the amount (Ac) in the central compartment when the drug is absorbed with a first-order or zero-order absorption process. &lt;br /&gt;
&amp;lt;blockquote&amp;gt;&lt;br /&gt;
'''Left:''' In the right-hand side window, the two (first-order and zero-order) models are described using the MLXtran coding language. In the left-hand side window, the structural model, experimental design, parameters and requested graphical output are defined. &lt;br /&gt;
&amp;lt;br&amp;gt;&amp;lt;br&amp;gt;&lt;br /&gt;
'''Right:''' The graphical output of MLXPlore, which was told to output the amount Ac in the central compartment with respect to time for zero-order (red) and first-order (blue) absorption.&lt;br /&gt;
&amp;lt;/blockquote&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellpadding=&amp;quot;10&amp;quot; cellspacing=&amp;quot;10&amp;quot;&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;| &lt;br /&gt;
[[File:Absorption1a_script.png]]&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption1a_bis.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
See  the [[Introduction to PK modeling using MLXPlore - Part II | Part II]]  for the corresponding amounts in the depot compartment and the related $\mlxplore$ project.&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== First-order, zero-order and $\alpha$-order absorption ===&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|absorption2a_script:}}&amp;lt;/span&amp;gt; we compute and display the amounts in the central and depot compartments when the drug is transferred from the depot to the central compartment with a first-order, zero-order or $\alpha$-order absorption process. &lt;br /&gt;
&lt;br /&gt;
Note $\dot{A}d(t) \, = \, -ka \, \times \, Ad^{\alpha}(t).$ Zero-order absorption is obtained with $\alpha=0$ and first-order absorption with $\alpha=1$. &lt;br /&gt;
The green curves are with respect to the $\alpha$-order absorption process.  &lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellspacing=&amp;quot;10&amp;quot; cellpadding=&amp;quot;10&amp;quot; &lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption2a_script.png]]&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption2a.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
See  the [[ Introduction to PK modeling using MLXPlore - Part II | Part II]]  for the corresponding amounts in the depot compartment and the related $\mlxplore$ project.  &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== First-order, zero-order and sequential zero-order/first-order absorption ===&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|absorption3a_script:}}&amp;lt;/span&amp;gt; we compute and display the amount in the central compartment when the drug is transferred from the depot to the central compartment with a first-order, zero-order or sequential zero-order/first-order absorption process. &lt;br /&gt;
&lt;br /&gt;
Here, $r0$ is the absorption rate for the zero-order process and $F0$ the fraction of the dose absorbed in a zero-order process.&lt;br /&gt;
The green curves refer to the sequential zero-order/first-order absorption process. &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellspacing=&amp;quot;10&amp;quot; cellpadding=&amp;quot;10&amp;quot;&lt;br /&gt;
|style=&amp;quot;width=50%&amp;quot;|&lt;br /&gt;
[[File:Absorption3a_script.png]]&lt;br /&gt;
|style=&amp;quot;width=50%|&lt;br /&gt;
[[File:Absorption3a.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
See  the [[ Introduction to PK modeling using MLXPlore - Part II | Part II]]  for the corresponding amounts in the depot compartment and the related $\mlxplore$ project.  &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== First-order and saturated absorption ===&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|absorption4_script:}}&amp;lt;/span&amp;gt; we compute and display the amount in the central compartment when the drug is transferred from the depot to the central compartment with a first-order or saturated (Michaelis-Mentens) absorption process.&lt;br /&gt;
The red curve is now for the saturated absorption process. &lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellspacing=&amp;quot;10&amp;quot; cellpadding=&amp;quot;10&amp;quot;&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption4a_script.png]]&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption4a.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
See  the [[ Introduction to PK modeling using MLXPlore - Part II| Part II]]  for the corresponding amounts in the depot compartment and the related $\mlxplore$ project.&lt;br /&gt;
&lt;br /&gt;
=== Lag-time and transit compartments ===&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|absorption5_script:}}&amp;lt;/span&amp;gt; we compute and display the amount in the central compartment when a lag time or a transit compartment model is used. &lt;br /&gt;
&lt;br /&gt;
Here, the blue curve is for first-order absorption without lag-time, the red curve for the lag-time model and the green one for the transit compartment model. The number of transit compartments is $Ntr=Mtt/Ktr$. When $Mtt=Tlag$, the transit compartment model can be seen as a smooth version of the lag-time model. It converges to the lag-time model when the number of compartments increases  (i.e., when the transfer rate constant $Ktr$ increases). &lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellspacing=&amp;quot;10&amp;quot; cellpadding=&amp;quot;10&amp;quot;&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption5a_script.png]]&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption5a.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
See  the [[ Introduction to PK modeling using MLXPlore - Part II| Part II]]  for the corresponding amounts in the depot compartment and the related $\mlxplore$ project.  &lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
=== Summary ===&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|absorption6a_script:}}&amp;lt;/span&amp;gt; we compute and display the amount in the central compartment for all of the different absorption models presented in the previous examples.&lt;br /&gt;
&lt;br /&gt;
In the figure, abs1 is first-order absorption, abs2 is $\alpha$-order absorption, abs3 is saturated absorption, abs4 is zero-order absorption and abs5 is sequential zero-order/first-order absorption.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellspacing=&amp;quot;10&amp;quot; cellpadding=&amp;quot;10&amp;quot;&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption6a_script.png]]&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Absorption6a.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
See  the [[ Introduction to PK modeling using MLXPlore - Part II | Part II]]  for the corresponding amounts in the depot compartment and the related $\mlxplore$ project.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Distribution ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
===  One, two and three compartment models ===&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|distribution1_script:}}&amp;lt;/span&amp;gt; we compute and display the amount in the central and peripheral compartments when the drug is distributed assuming one, two or three compartment models.&lt;br /&gt;
&lt;br /&gt;
Here, $Ap$ and $Aq$ are the amounts in the first and second peripheral compartments and $lAc$ the log-amount in the central compartment. &lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellspacing=&amp;quot;10&amp;quot; cellpadding=&amp;quot;10&amp;quot;&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Distribution1_script.png]]&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Distribution1.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
&lt;br /&gt;
== Elimination ==&lt;br /&gt;
&lt;br /&gt;
&amp;lt;br&amp;gt;&lt;br /&gt;
=== Linear, nonlinear and combined elimination ===&lt;br /&gt;
&lt;br /&gt;
&amp;lt;span style=&amp;quot;color:#993300&amp;quot;&amp;gt;{{Verbatim|elimination1_script:}}&amp;lt;/span&amp;gt; we compute and display the amount in the central compartment and the rate of elimination when the drug is eliminated with a linear, nonlinear (Michaelis-Mentens) or combined elimination process (linear when $\alpha=1$ and Michaelis-Mentens when $\alpha=1$).&lt;br /&gt;
&lt;br /&gt;
Here, $lAc$ is the log-amount in the central compartment and lre the log-rate of elimination of the drug. By definition, lre is a linear function of time for a linear elimination process.&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&amp;lt;div style=&amp;quot;overflow-x:auto&amp;quot;&amp;gt;&lt;br /&gt;
{| cellspacing=&amp;quot;10&amp;quot; cellpadding=&amp;quot;10&amp;quot;&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Elimination1_script.png|460px]]&lt;br /&gt;
|style=&amp;quot;width:50%&amp;quot;|&lt;br /&gt;
[[File:Plot_Elimination1.png]]&lt;br /&gt;
|} &amp;lt;/div&amp;gt;&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
&lt;br /&gt;
All the projects shown in this session can be downloaded here: {{filepath:Pk mlxplore.zip}}.&lt;br /&gt;
&lt;br /&gt;
{{Next&lt;br /&gt;
|link=Introduction to PK modeling using MLXPlore - Part II }}&lt;/div&gt;</summary>
		<author><name>Admin</name></author>
		
	</entry>
</feed>