<?xml version="1.0"?>
<feed xmlns="http://www.w3.org/2005/Atom" xml:lang="en">
	<id>https://calculus.subwiki.org/w/index.php?action=history&amp;feed=atom&amp;title=Logistic_log-loss_function_of_one_variable</id>
	<title>Logistic log-loss function of one variable - Revision history</title>
	<link rel="self" type="application/atom+xml" href="https://calculus.subwiki.org/w/index.php?action=history&amp;feed=atom&amp;title=Logistic_log-loss_function_of_one_variable"/>
	<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;action=history"/>
	<updated>2026-08-12T13:15:21Z</updated>
	<subtitle>Revision history for this page on the wiki</subtitle>
	<generator>MediaWiki 1.41.2</generator>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3071&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3071&amp;oldid=prev"/>
		<updated>2014-09-28T02:02:29Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 02:02, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l114&quot;&gt;Line 114:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 114:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;{| class=&amp;quot;sortable&amp;quot; border=&amp;quot;1&amp;quot;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;{| class=&amp;quot;sortable&amp;quot; border=&amp;quot;1&amp;quot;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;(general case) &lt;/ins&gt;!! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear &lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;(unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case &lt;/del&gt;convergence &lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;is quadratic. &lt;/del&gt;|| &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)| = |1 - \alpha p(1- p)|&amp;lt;/math&amp;gt;&amp;lt;br&amp;gt;If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;\left| \frac{1}{2} - p \right|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear convergence|| &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)| = |1 - \alpha p(1- p)|&amp;lt;/math&amp;gt;&amp;lt;br&amp;gt;If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;\left| \frac{1}{2} - p \right|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3069&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3069&amp;oldid=prev"/>
		<updated>2014-09-28T01:43:22Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 01:43, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l116&quot;&gt;Line 116:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 116:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)| = |1 - \alpha p(1- p)&amp;lt;/math&amp;gt; &lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;if &lt;/del&gt;&amp;lt;&lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&lt;/del&gt;&amp;gt;&lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;. &lt;/del&gt;If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;\left| \frac{1}{2} - p \right|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)| = |1 - \alpha p(1- p)&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;|&lt;/ins&gt;&amp;lt;/math&amp;gt;&amp;lt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;br&lt;/ins&gt;&amp;gt;If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;\left| \frac{1}{2} - p \right|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3068&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3068&amp;oldid=prev"/>
		<updated>2014-09-28T01:41:40Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 01:41, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l116&quot;&gt;Line 116:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 116:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)| = |1 - \alpha p(1- p)&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;\left| \frac{1}{2} - p \right|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)| = |1 - \alpha p(1- p)&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;\left| \frac{1}{2} - p \right|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;cubic convergence with convergence rate &lt;/ins&gt;&amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3067&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3067&amp;oldid=prev"/>
		<updated>2014-09-28T01:41:16Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 01:41, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l116&quot;&gt;Line 116:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 116:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;|1 - &lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;2p&lt;/del&gt;|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)| &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;= |1 - \alpha p(1- p)&lt;/ins&gt;&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;\left&lt;/ins&gt;| &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;\frac{&lt;/ins&gt;1&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;}{2} &lt;/ins&gt;- &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;p \right&lt;/ins&gt;|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3066&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3066&amp;oldid=prev"/>
		<updated>2014-09-28T01:39:48Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 01:39, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l116&quot;&gt;Line 116:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 116:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate (general case) !! Convergence rate (special cases)&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;|1 - 2p|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;/&lt;/ins&gt;math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;|1 - 2p|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&amp;#039;s method for optimization of a logistic log-loss function of one variable]] || &amp;#039;&amp;#039;Not&amp;#039;&amp;#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2}&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3065&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3065&amp;oldid=prev"/>
		<updated>2014-09-28T01:36:18Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 01:36, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l118&quot;&gt;Line 118:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 118:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&amp;#039;&amp;#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&amp;#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&amp;#039;&amp;#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&amp;#039;&amp;#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&amp;#039;&amp;#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;|1 - 2p|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&amp;#039;&amp;#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&amp;#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&amp;#039;&amp;#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&amp;#039;&amp;#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&amp;#039;&amp;#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. || Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;|1 - 2p|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&amp;#039;&amp;#039;&amp;#039;(0)}{g&amp;#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&#039;s method for optimization of a logistic log-loss function of one variable]] || &#039;&#039;Not&#039;&#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2&lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;]&lt;/del&gt;&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&#039;s method for optimization of a logistic log-loss function of one variable]] || &#039;&#039;Not&#039;&#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;}&lt;/ins&gt;&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{6}&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3064&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3064&amp;oldid=prev"/>
		<updated>2014-09-28T01:36:04Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 01:36, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l114&quot;&gt;Line 114:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 114:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;{| class=&amp;quot;sortable&amp;quot; border=&amp;quot;1&amp;quot;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;{| class=&amp;quot;sortable&amp;quot; border=&amp;quot;1&amp;quot;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;(general case) !! Convergence rate (special cases)&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;.&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]], with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;. &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;|| Case &amp;lt;math&amp;gt;\alpha = \frac{1}{p(1 - p)}&amp;lt;math&amp;gt; and &amp;lt;math&amp;gt;p \ne \frac{1}{2}&amp;lt;/math&amp;gt;: quadratic convergence with convergence rate &amp;lt;math&amp;gt;|1 - 2p|&amp;lt;/math&amp;gt;.&amp;lt;br&amp;gt;Case that &amp;lt;math&amp;gt;p = \frac{1}{2}, \alpha = 4&amp;lt;/math&amp;gt;: &amp;lt;math&amp;gt;\frac{1}{6} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{12}&amp;lt;/math&amp;gt;&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&#039;s method for optimization of a logistic log-loss function of one variable]] || &#039;&#039;Not&#039;&#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[Newton&#039;s method for optimization of a logistic log-loss function of one variable]] || &#039;&#039;Not&#039;&#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;&amp;lt;/math&amp;gt; || Case that &amp;lt;math&amp;gt;p = \frac{1}{2]&amp;lt;/math&amp;gt;: cubic convergence with convergence rate &amp;lt;math&amp;gt;\frac{1}{3} \left| \frac{g&#039;&#039;&#039;(0)}{g&#039;(0)} \right| = \frac{1}{6}&lt;/ins&gt;&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3059&amp;oldid=prev</id>
		<title>Vipul: /* Optimization methods */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3059&amp;oldid=prev"/>
		<updated>2014-09-28T01:09:10Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Optimization methods&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 01:09, 28 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l111&quot;&gt;Line 111:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 111:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;==Optimization methods==&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;==Optimization methods==&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;* [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]]&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;We will denote the point of minimum, &amp;lt;math&amp;gt;g^{-1}(p)&amp;lt;/math&amp;gt;, as &amp;lt;math&amp;gt;x^&lt;/ins&gt;*&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;&amp;lt;/math&amp;gt;.&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;* &lt;/del&gt;[[Newton&#039;s method for optimization of a logistic log-loss function of one variable]]&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt; &lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;{| class=&quot;sortable&quot; border=&quot;1&quot;&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;! Method !! Domain of convergence !! Order of convergence !! Convergence rate&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;|-&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;| &lt;/ins&gt;[[Gradient descent with constant learning rate for a logistic log-loss function of one variable]]&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;, with learning rate &amp;lt;math&amp;gt;\alpha&amp;lt;/math&amp;gt; || If &amp;lt;math&amp;gt;\alpha &amp;lt; \frac{1}{f&#039;&#039;(x^*)} = \frac{1}{p(1 - p)}&amp;lt;/math&amp;gt; with &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt; the point of minimum, convergence occurs from any starting point, but can be quite slow. Since we don&#039;t know &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, setting &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt; is a safe choice. || Linear (unless &amp;lt;math&amp;gt;\alpha = \frac{1}{f&#039;&#039;(x*)}&amp;lt;/math&amp;gt;, in which case convergence is quadratic. || &amp;lt;math&amp;gt;|1 - \alpha f&#039;&#039;(x^*)|&amp;lt;/math&amp;gt; if &amp;lt;math&amp;gt;\alpha \ne \frac{1}{f&#039;&#039;(x^*)}&amp;lt;/math&amp;gt;. If we set &amp;lt;math&amp;gt;\alpha = 4&amp;lt;/math&amp;gt;, we get &amp;lt;math&amp;gt;|1 - 4p(1 - p)|&amp;lt;/math&amp;gt;.&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;|-&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;| &lt;/ins&gt;[[Newton&#039;s method for optimization of a logistic log-loss function of one variable]] &lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;|| &#039;&#039;Not&#039;&#039; all reals. However, the domain includes the interval between 0 and &amp;lt;math&amp;gt;x^*&amp;lt;/math&amp;gt;, and convergence is monotone on the interval. || Quadratic convergence. || &amp;lt;math&amp;gt;\left|\frac{1}{2} - p \right|&amp;lt;/math&amp;gt;&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;|}&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3055&amp;oldid=prev</id>
		<title>Vipul at 05:04, 22 September 2014</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3055&amp;oldid=prev"/>
		<updated>2014-09-22T05:04:22Z</updated>

		<summary type="html">&lt;p&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 05:04, 22 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l108&quot;&gt;Line 108:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 108:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;This is always positive, so the function is a convex function and its graph is concave up everywhere. In particular, there are no points of inflection.&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;This is always positive, so the function is a convex function and its graph is concave up everywhere. In particular, there are no points of inflection.&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;==Optimization methods==&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;* [[Gradient descent with constant learning rate for a logistic log-loss function of one variable]]&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-side-deleted&quot;&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;* [[Newton&#039;s method for optimization of a logistic log-loss function of one variable]]&lt;/ins&gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
	<entry>
		<id>https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3049&amp;oldid=prev</id>
		<title>Vipul: /* Key data */</title>
		<link rel="alternate" type="text/html" href="https://calculus.subwiki.org/w/index.php?title=Logistic_log-loss_function_of_one_variable&amp;diff=3049&amp;oldid=prev"/>
		<updated>2014-09-22T04:50:37Z</updated>

		<summary type="html">&lt;p&gt;&lt;span dir=&quot;auto&quot;&gt;&lt;span class=&quot;autocomment&quot;&gt;Key data&lt;/span&gt;&lt;/span&gt;&lt;/p&gt;
&lt;table style=&quot;background-color: #fff; color: #202122;&quot; data-mw=&quot;interface&quot;&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;col class=&quot;diff-marker&quot; /&gt;
				&lt;col class=&quot;diff-content&quot; /&gt;
				&lt;tr class=&quot;diff-title&quot; lang=&quot;en&quot;&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;← Older revision&lt;/td&gt;
				&lt;td colspan=&quot;2&quot; style=&quot;background-color: #fff; color: #202122; text-align: center;&quot;&gt;Revision as of 04:50, 22 September 2014&lt;/td&gt;
				&lt;/tr&gt;&lt;tr&gt;&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot; id=&quot;mw-diff-left-l36&quot;&gt;Line 36:&lt;/td&gt;
&lt;td colspan=&quot;2&quot; class=&quot;diff-lineno&quot;&gt;Line 36:&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[second derivative]] || &amp;lt;math&amp;gt;g(x)(1 - g(x)) = g(x)g(-x)&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[second derivative]] || &amp;lt;math&amp;gt;g(x)(1 - g(x)) = g(x)g(-x)&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|-&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;−&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #ffe49c; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[third derivative]] || &amp;lt;math&amp;gt;g(x)g(-x)(g(x) - g(&lt;del style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;-&lt;/del&gt;x)) = g(x)(1 - g(x))(1 - 2g(x))&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot; data-marker=&quot;+&quot;&gt;&lt;/td&gt;&lt;td style=&quot;color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #a3d3ff; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;| [[third derivative]] || &amp;lt;math&amp;gt;g(x)g(-x)(g(&lt;ins style=&quot;font-weight: bold; text-decoration: none;&quot;&gt;-&lt;/ins&gt;x) - g(x)) = g(x)(1 - g(x))(1 - 2g(x))&amp;lt;/math&amp;gt;&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;div&gt;|}&lt;/div&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;tr&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;td class=&quot;diff-marker&quot;&gt;&lt;/td&gt;&lt;td style=&quot;background-color: #f8f9fa; color: #202122; font-size: 88%; border-style: solid; border-width: 1px 1px 1px 4px; border-radius: 0.33em; border-color: #eaecf0; vertical-align: top; white-space: pre-wrap;&quot;&gt;&lt;br&gt;&lt;/td&gt;&lt;/tr&gt;
&lt;/table&gt;</summary>
		<author><name>Vipul</name></author>
	</entry>
</feed>